File size: 5,845 Bytes
07f059a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
from __future__ import annotations
"""
์ง์žฅ์ธ(worker-v3) ๋Œ€๊ทœ๋ชจ ๋žœ๋ค ์ŠคํŠธ๋ ˆ์Šค ์‹œ๋ฎฌ๋ ˆ์ด์…˜ โ€” ์˜๋„ ๋ณ€ํ™” ์ ์ ˆ์„ฑยท์ด์ƒ ์ผ€์ด์Šค ํƒ์ง€.

worker๋Š” single-select(์•ฑ app_open). intent 9๊ฐœ๋ผ ๋ฐ˜์‘์„ฑ์€ top-3 ๊ธฐ์ค€.
์ง„ํ–‰๋ฐ” /tmp/worker_sim_progress.txt, ์š”์•ฝ stdout, ์ƒ์„ธ /tmp/worker_sim_report.json.

ํƒ์ง€: E1 ๋ฒ”์œ„/NaN, E3 ์ƒ์œ„ ๋™์ , E4 ํ–‰๋™๋ฐ˜์‘(top-3), E5 stuck + ๋ถ„ํฌ๊ฑด์ „์„ฑ/์•ฑ๋ณ„ ๋ฐ˜์‘๋ฅ .
(E2 ๋ฒ”์šฉ intent ๋ˆ„์ถœ์€ worker์— ๋ฒ”์šฉ intent ์—†์–ด N/A)
"""
import json, random, sys, time
from collections import Counter, defaultdict
from pathlib import Path

sys.path.insert(0, str(Path(__file__).parent.parent))
from core.engines import config                              # noqa: E402
from core.extractor import get_extractor                     # noqa: E402
from core.inference import infer_batch, infer_with_behavior  # noqa: E402

SID = "worker-v3"
N_SURVEY = 50_000
N_SESS = 8_000
MAX_ACT = 30
STUCK_N = 15
TOPK_RESP = 3            # 9 intent โ†’ top-3 ๊ธฐ์ค€ ๋ฐ˜์‘์„ฑ
PROG = "/tmp/worker_sim_progress.txt"
REPORT = "/tmp/worker_sim_report.json"

tax = config.get_taxonomy(SID)["intents"]
NM = {i["id"]: i["name"] for i in tax}
SURVEY = config.get_survey(SID)["questions"]
OPTS = {q["id"]: [o["code"] for o in q["options"]] for q in SURVEY}
SIG = config.get_behavior_signals(SID)
APPS = config.get_behaviors(SID)["apps"]


def _bar(frac, label):
    n = int(frac * 30)
    with open(PROG, "w") as f:
        f.write(f"[{'#'*n}{'.'*(30-n)}] {frac*100:5.1f}%  {label}")


def _rand_survey(rng):
    return {q: rng.choice(OPTS[q]) for q in OPTS}


def _top(scores, k):
    return sorted(scores, key=lambda s: s.final_score, reverse=True)[:k]


def _check_common(scores, anom, tag):
    for s in scores:
        v = s.final_score
        if v != v or v < -1e-9 or v > 1.0001:
            anom["E1"].append((tag, s.intent_id, v)); break
    top = _top(scores, 5)
    top_ids = [s.intent_id for s in top]
    tv = round(top[0].final_score, 2)
    tie = sum(1 for s in top if round(s.final_score, 2) == tv)
    if tie >= 3:
        anom["E3"].append((tag, tv, tie, top_ids[:tie]))
    return top_ids


def run():
    ext = get_extractor()
    rng = random.Random(20240601)
    t0 = time.time()
    anom = defaultdict(list)
    top1_survey = Counter()
    for i in range(N_SURVEY):
        a = _rand_survey(rng)
        try:
            _, sc = infer_batch(a, SID)
        except Exception as e:
            anom["E1"].append(("survey", "EXCEPTION", str(e)[:80])); continue
        top1_survey[_check_common(sc, anom, "survey")[0]] += 1
        if i % 1000 == 0:
            _bar(i / (N_SURVEY + N_SESS), f"์„ค๋ฌธ {i:,}/{N_SURVEY:,}")
    app_rise = defaultdict(lambda: [0, 0])
    stuck_runs = []
    for j in range(N_SESS):
        a = _rand_survey(rng)
        sess = f"__ws_{j}"
        ext.reset(sess)
        last_top1 = None; run_len = 0; max_run = 0
        for _ in range(rng.randint(5, MAX_ACT)):
            app = rng.choice(APPS)
            en = app["entity"]
            ext.add_event(sess, app.get("event_type", "app_open"), en)
            try:
                _, sc = infer_with_behavior(a, sess, SID)
            except Exception as e:
                anom["E1"].append((f"beh_{j}", "EXCEPTION", str(e)[:80])); break
            top_ids = _check_common(sc, anom, f"beh_{j}")
            targets = SIG.get(en, [])
            if targets:
                app_rise[en][1] += 1
                if set(targets) & set([s.intent_id for s in _top(sc, TOPK_RESP)]):
                    app_rise[en][0] += 1
            if top_ids[0] == last_top1:
                run_len += 1; max_run = max(max_run, run_len)
            else:
                last_top1 = top_ids[0]; run_len = 1
        if max_run >= STUCK_N:
            stuck_runs.append((f"beh_{j}", last_top1, max_run))
        ext.reset(sess)
        if j % 100 == 0:
            _bar((N_SURVEY + j) / (N_SURVEY + N_SESS), f"ํ–‰๋™ {j:,}/{N_SESS:,}")
    _bar(1.0, "์™„๋ฃŒ")
    dt = time.time() - t0

    n = sum(top1_survey.values())
    hhi = sum((c / n) ** 2 for c in top1_survey.values()) if n else 0
    low = {e: f"{v[0]}/{v[1]}={v[0]/v[1]*100:.0f}%" for e, v in app_rise.items() if v[1] and v[0]/v[1] < 0.9}
    summary = {
        "elapsed_sec": round(dt, 1), "n_survey": N_SURVEY, "n_sess": N_SESS, "topk_resp": TOPK_RESP,
        "E1_range_nan": len(anom["E1"]), "E3_top_tie(>=3)": len(anom["E3"]),
        "E5_stuck(>=%d)" % STUCK_N: len(stuck_runs),
        "survey_top1_distinct": len(top1_survey), "survey_top1_HHI": round(hhi, 4),
        "survey_top1": [(NM[i][:16], f"{c/n*100:.1f}%") for i, c in top1_survey.most_common()],
        "app_low_responsiveness(<90%)": low,
        "stuck_samples": stuck_runs[:10], "E3_samples": anom["E3"][:5], "E1_samples": anom["E1"][:5],
    }
    json.dump({"summary": summary, "anom_E1": anom["E1"][:200],
               "anom_E3": anom["E3"][:300], "stuck": stuck_runs[:300]},
              open(REPORT, "w"), ensure_ascii=False, indent=1)

    print("=" * 64)
    print(f"์ง์žฅ์ธ ์ŠคํŠธ๋ ˆ์Šค ์‹œ๋ฎฌ ์™„๋ฃŒ โ€” {dt/60:.1f}๋ถ„ (์„ค๋ฌธ {N_SURVEY:,} + ํ–‰๋™ {N_SESS:,}ร—โ‰ค{MAX_ACT})")
    print("=" * 64)
    print(f"  [E1] ๋ฒ”์œ„/NaN/์˜ˆ์™ธ      : {len(anom['E1'])} ๊ฑด")
    print(f"  [E3] ์ƒ์œ„ ๋™์ (โ‰ฅ3)      : {len(anom['E3'])} ๊ฑด")
    print(f"  [E5] stuck(top1 โ‰ฅ{STUCK_N}์—ฐ์†): {len(stuck_runs)} ์„ธ์…˜")
    print(f"  ์„ค๋ฌธ top-1 ๋‹ค์–‘์„ฑ: {len(top1_survey)}/9์ข…, HHI={hhi:.3f}")
    print("  ์„ค๋ฌธ top-1: " + ", ".join(f"{NM[i][:8]} {c/n*100:.0f}%" for i, c in top1_survey.most_common(6)))
    print(f"  ์•ฑ ๋ฐ˜์‘๋ฅ (<90%, top-{TOPK_RESP}): {low if low else '์—†์Œ โœ“'}")
    if stuck_runs:
        print("  stuck ์ƒ˜ํ”Œ:", [(NM[s[1]][:10], s[2]) for s in stuck_runs[:5]])
    print(f"\n์ƒ์„ธ: {REPORT}")


if __name__ == "__main__":
    run()