#!/usr/bin/env python3 # -*- coding: utf-8 -*- """exposure_expected.py — تقدير التعرّض بلا افتراض إعادة بناء مطابقة: - معدل سحب كل صف مجمّع إلى الأمثلة = 119,961/3,608,654 = 3.32% (من إعادة تشغيل RNG) - نعدّ لكل عنصر تقييم: كم صف مجمّع يحتوي نصه (المطابق تطبيعًا) على أي من الجانبين ⇒ احتمال تعرّضه = 1-(1-p)^k (k صفوف تحمل النص نفسه) - يقارن مع العدد المُحقَّق من إعادة البناء (exposure_analysis.json).""" import json, hashlib, sys, time import numpy as np sys.path.insert(0,"/home/user/lhc/recon") exec(open('/home/user/lhc/recon/build_masks_stream.py').read().split('# ---------------- مجموعات التقييم')[0]) from phase_w1_data import download_corpus T0=time.time() def log(*a): print(f"[{time.time()-T0:6.1f}s]",*a,flush=True) P_SAMPLE = 119961/3608654 log(f"معدل السحب لكل صف: {P_SAMPLE*100:.3f}%") DD="/home/user/lhc/data/opus" OPUS=[("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000), ("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000), ("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)] pairs={} for name,url,lim in OPUS: pairs[name]=download_corpus(name,url,lim,DD) DEV="/home/user/lhc/data/flores101_dataset/devtest" ld=lambda p:[l.strip() for l in open(p,encoding="utf-8") if l.strip()] pairs["FLORES"]=[{"ar":a,"en":e} for a,e in zip(ld(f"{DEV}/ara.devtest"),ld(f"{DEV}/eng.devtest"))] def h(s): return int.from_bytes(hashlib.md5(s.encode()).digest()[:8], "little") targets={} # hash → (set, item) for cn,ps in pairs.items(): for i,p in enumerate(ps): targets[h(norm_ar(p["ar"]))]=(cn,i) targets[h(norm_en(p["en"]))]=(cn,i) log(f"أهداف المطابقة: {len(targets):,} بصمة") counts={cn: np.zeros(len(ps), dtype=np.int32) for cn,ps in pairs.items()} with open("/var/tmp/lhc_w4/pairs.jsonl", encoding="utf-8") as f: for n,line in enumerate(f): try: o=json.loads(line) except Exception: continue ar,en=o.get("ar",""),o.get("en","") if ar: t=targets.get(h(norm_ar(ar))) if t: counts[t[0]][t[1]]+=1 if en: t=targets.get(h(norm_en(en))) if t: counts[t[0]][t[1]]+=1 if (n+1)%1000000==0: log(f" مسح {n+1:,}") log("انتهى المسح") rep={"p_sample":round(P_SAMPLE,6),"sets":{}} recon=json.load(open("/home/user/lhc/results/exposure_analysis.json")) for cn,ps in pairs.items(): k=counts[cn] exp=float(np.sum(1-(1-P_SAMPLE)**k)) rep["sets"][cn]={"n":len(ps), "rows_with_text_total":int(k.sum()), "items_with_k_ge1":int((k>=1).sum()), "expected_exposed":round(exp,1), "expected_pct":round(100*exp/len(ps),3), "realized_recon_exposed":recon["sets"][cn]["exposed_exact"], "k_distribution":{str(int(v)):int((k==v).sum()) for v in np.unique(k) if v>0 and v<=10}} log(f"{cn:12s} صفوف تحمل نصوص العناصر={int(k.sum()):,} | عناصر k≥1={int((k>=1).sum()):,} | متوقع= {exp:.1f} | محقق(إعادة بناء)={recon['sets'][cn]['exposed_exact']}") json.dump(rep, open("/home/user/lhc/results/exposure_expected.json","w"), ensure_ascii=False, indent=1) log("SAVED results/exposure_expected.json")