File size: 3,484 Bytes
9074b68
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""exposure_expected.py — تقدير التعرّض بلا افتراض إعادة بناء مطابقة:
- معدل سحب كل صف مجمّع إلى الأمثلة = 119,961/3,608,654 = 3.32% (من إعادة تشغيل RNG)
- نعدّ لكل عنصر تقييم: كم صف مجمّع يحتوي نصه (المطابق تطبيعًا) على أي من الجانبين
  ⇒ احتمال تعرّضه = 1-(1-p)^k (k صفوف تحمل النص نفسه)
- يقارن مع العدد المُحقَّق من إعادة البناء (exposure_analysis.json)."""
import json, hashlib, sys, time
import numpy as np
sys.path.insert(0,"/home/user/lhc/recon")
exec(open('/home/user/lhc/recon/build_masks_stream.py').read().split('# ---------------- مجموعات التقييم')[0])
from phase_w1_data import download_corpus
T0=time.time()
def log(*a): print(f"[{time.time()-T0:6.1f}s]",*a,flush=True)
P_SAMPLE = 119961/3608654
log(f"معدل السحب لكل صف: {P_SAMPLE*100:.3f}%")

DD="/home/user/lhc/data/opus"
OPUS=[("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000),
      ("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000),
      ("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)]
pairs={}
for name,url,lim in OPUS: pairs[name]=download_corpus(name,url,lim,DD)
DEV="/home/user/lhc/data/flores101_dataset/devtest"
ld=lambda p:[l.strip() for l in open(p,encoding="utf-8") if l.strip()]
pairs["FLORES"]=[{"ar":a,"en":e} for a,e in zip(ld(f"{DEV}/ara.devtest"),ld(f"{DEV}/eng.devtest"))]

def h(s): return int.from_bytes(hashlib.md5(s.encode()).digest()[:8], "little")
targets={}   # hash → (set, item)
for cn,ps in pairs.items():
    for i,p in enumerate(ps):
        targets[h(norm_ar(p["ar"]))]=(cn,i)
        targets[h(norm_en(p["en"]))]=(cn,i)
log(f"أهداف المطابقة: {len(targets):,} بصمة")
counts={cn: np.zeros(len(ps), dtype=np.int32) for cn,ps in pairs.items()}
with open("/var/tmp/lhc_w4/pairs.jsonl", encoding="utf-8") as f:
    for n,line in enumerate(f):
        try: o=json.loads(line)
        except Exception: continue
        ar,en=o.get("ar",""),o.get("en","")
        if ar:
            t=targets.get(h(norm_ar(ar)))
            if t: counts[t[0]][t[1]]+=1
        if en:
            t=targets.get(h(norm_en(en)))
            if t: counts[t[0]][t[1]]+=1
        if (n+1)%1000000==0: log(f"  مسح {n+1:,}")
log("انتهى المسح")
rep={"p_sample":round(P_SAMPLE,6),"sets":{}}
recon=json.load(open("/home/user/lhc/results/exposure_analysis.json"))
for cn,ps in pairs.items():
    k=counts[cn]
    exp=float(np.sum(1-(1-P_SAMPLE)**k))
    rep["sets"][cn]={"n":len(ps),
        "rows_with_text_total":int(k.sum()),
        "items_with_k_ge1":int((k>=1).sum()),
        "expected_exposed":round(exp,1),
        "expected_pct":round(100*exp/len(ps),3),
        "realized_recon_exposed":recon["sets"][cn]["exposed_exact"],
        "k_distribution":{str(int(v)):int((k==v).sum()) for v in np.unique(k) if v>0 and v<=10}}
    log(f"{cn:12s} صفوف تحمل نصوص العناصر={int(k.sum()):,} | عناصر k≥1={int((k>=1).sum()):,} | متوقع= {exp:.1f}  | محقق(إعادة بناء)={recon['sets'][cn]['exposed_exact']}")
json.dump(rep, open("/home/user/lhc/results/exposure_expected.json","w"), ensure_ascii=False, indent=1)
log("SAVED results/exposure_expected.json")