Download code/exposure_expected.py from sayed125/lhc-0-brain: direct link, hf CLI and curl.
- Browser
- Download file 3.48 kB
-
https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/exposure_expected.py
- Command line
-
hf download hf://sayed125/lhc-0-brain/code/exposure_expected.py
-
curl -L -o exposure_expected.py https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/exposure_expected.py
3.48 kB
| #!/usr/bin/env python3 | |
| # -*- coding: utf-8 -*- | |
| """exposure_expected.py — تقدير التعرّض بلا افتراض إعادة بناء مطابقة: | |
| - معدل سحب كل صف مجمّع إلى الأمثلة = 119,961/3,608,654 = 3.32% (من إعادة تشغيل RNG) | |
| - نعدّ لكل عنصر تقييم: كم صف مجمّع يحتوي نصه (المطابق تطبيعًا) على أي من الجانبين | |
| ⇒ احتمال تعرّضه = 1-(1-p)^k (k صفوف تحمل النص نفسه) | |
| - يقارن مع العدد المُحقَّق من إعادة البناء (exposure_analysis.json).""" | |
| import json, hashlib, sys, time | |
| import numpy as np | |
| sys.path.insert(0,"/home/user/lhc/recon") | |
| exec(open('/home/user/lhc/recon/build_masks_stream.py').read().split('# ---------------- مجموعات التقييم')[0]) | |
| from phase_w1_data import download_corpus | |
| T0=time.time() | |
| def log(*a): print(f"[{time.time()-T0:6.1f}s]",*a,flush=True) | |
| P_SAMPLE = 119961/3608654 | |
| log(f"معدل السحب لكل صف: {P_SAMPLE*100:.3f}%") | |
| DD="/home/user/lhc/data/opus" | |
| OPUS=[("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000), | |
| ("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000), | |
| ("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)] | |
| pairs={} | |
| for name,url,lim in OPUS: pairs[name]=download_corpus(name,url,lim,DD) | |
| DEV="/home/user/lhc/data/flores101_dataset/devtest" | |
| ld=lambda p:[l.strip() for l in open(p,encoding="utf-8") if l.strip()] | |
| pairs["FLORES"]=[{"ar":a,"en":e} for a,e in zip(ld(f"{DEV}/ara.devtest"),ld(f"{DEV}/eng.devtest"))] | |
| def h(s): return int.from_bytes(hashlib.md5(s.encode()).digest()[:8], "little") | |
| targets={} # hash → (set, item) | |
| for cn,ps in pairs.items(): | |
| for i,p in enumerate(ps): | |
| targets[h(norm_ar(p["ar"]))]=(cn,i) | |
| targets[h(norm_en(p["en"]))]=(cn,i) | |
| log(f"أهداف المطابقة: {len(targets):,} بصمة") | |
| counts={cn: np.zeros(len(ps), dtype=np.int32) for cn,ps in pairs.items()} | |
| with open("/var/tmp/lhc_w4/pairs.jsonl", encoding="utf-8") as f: | |
| for n,line in enumerate(f): | |
| try: o=json.loads(line) | |
| except Exception: continue | |
| ar,en=o.get("ar",""),o.get("en","") | |
| if ar: | |
| t=targets.get(h(norm_ar(ar))) | |
| if t: counts[t[0]][t[1]]+=1 | |
| if en: | |
| t=targets.get(h(norm_en(en))) | |
| if t: counts[t[0]][t[1]]+=1 | |
| if (n+1)%1000000==0: log(f" مسح {n+1:,}") | |
| log("انتهى المسح") | |
| rep={"p_sample":round(P_SAMPLE,6),"sets":{}} | |
| recon=json.load(open("/home/user/lhc/results/exposure_analysis.json")) | |
| for cn,ps in pairs.items(): | |
| k=counts[cn] | |
| exp=float(np.sum(1-(1-P_SAMPLE)**k)) | |
| rep["sets"][cn]={"n":len(ps), | |
| "rows_with_text_total":int(k.sum()), | |
| "items_with_k_ge1":int((k>=1).sum()), | |
| "expected_exposed":round(exp,1), | |
| "expected_pct":round(100*exp/len(ps),3), | |
| "realized_recon_exposed":recon["sets"][cn]["exposed_exact"], | |
| "k_distribution":{str(int(v)):int((k==v).sum()) for v in np.unique(k) if v>0 and v<=10}} | |
| log(f"{cn:12s} صفوف تحمل نصوص العناصر={int(k.sum()):,} | عناصر k≥1={int((k>=1).sum()):,} | متوقع= {exp:.1f} | محقق(إعادة بناء)={recon['sets'][cn]['exposed_exact']}") | |
| json.dump(rep, open("/home/user/lhc/results/exposure_expected.json","w"), ensure_ascii=False, indent=1) | |
| log("SAVED results/exposure_expected.json") | |