File size: 5,426 Bytes
753201e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f2ca89c
753201e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
"""
evaluate.py
-----------
Evaluate the trained LightGBM model on the held-out split.

Outputs:
  eval/eval_results.json β€” precision, recall, F1, false-positive cost,
                            latency, and policy-action breakdown.

Usage:
    python eval/evaluate.py
"""

import json
import os
import sys
import time

import lightgbm as lgb
import numpy as np
from sklearn.metrics import precision_score, recall_score, f1_score

BASE_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, BASE_DIR)

SCORING_DIR = os.path.join(BASE_DIR, "scoring")
EVAL_DIR    = os.path.join(BASE_DIR, "eval")

MODEL_PATH   = os.path.join(SCORING_DIR, "model.lgb")
HOLDOUT_PATH = os.path.join(SCORING_DIR, "holdout_split.json")
RESULTS_PATH = os.path.join(EVAL_DIR, "eval_results.json")

CLASSIFICATION_THRESHOLD = 0.5

# Policy thresholds (must match agent/policy.py exactly)
POLICY_HIGH_THRESHOLD    = 0.85
POLICY_MEDIUM_THRESHOLD  = 0.50
POLICY_LOW_EXPOSURE      = 25000.0   # updated Phase 3 v3.0 β€” matches agent/policy.py


def policy_decide(risk_probability: float, exposure_rupees: float) -> str:
    """Deterministic policy gate β€” mirrors agent/policy.py exactly."""
    if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees <= POLICY_LOW_EXPOSURE:
        return "auto_hold"
    if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees > POLICY_LOW_EXPOSURE:
        return "escalate"
    if POLICY_MEDIUM_THRESHOLD <= risk_probability < POLICY_HIGH_THRESHOLD:
        return "escalate"
    return "log_only"


def evaluate():
    print(f"Loading model from {MODEL_PATH}...")
    model = lgb.Booster(model_file=MODEL_PATH)

    print(f"Loading holdout split from {HOLDOUT_PATH}...")
    with open(HOLDOUT_PATH, encoding="utf-8") as f:
        holdout = json.load(f)

    print(f"  {len(holdout)} held-out clusters")

    # 6.1 β€” reconstruct feature matrix
    from scoring.features import FEATURE_COLUMNS

    feature_rows = [row["features"] for row in holdout]
    y_true = np.array([row["label"] for row in holdout])

    # 6.2 β€” time the feature extraction + inference pass
    start = time.perf_counter()
    X = np.array([[fr[f] for f in FEATURE_COLUMNS] for fr in feature_rows])
    probs = model.predict(X)
    elapsed = time.perf_counter() - start

    y_pred = (probs >= CLASSIFICATION_THRESHOLD).astype(int)

    precision = float(precision_score(y_true, y_pred, zero_division=0))
    recall    = float(recall_score(y_true, y_pred, zero_division=0))
    f1        = float(f1_score(y_true, y_pred, zero_division=0))
    latency_total_ms   = elapsed * 1000
    latency_per_cluster_ms = latency_total_ms / len(holdout)

    print(f"\n--- Classification at threshold {CLASSIFICATION_THRESHOLD} ---")
    print(f"  Precision : {precision:.4f}")
    print(f"  Recall    : {recall:.4f}")
    print(f"  F1        : {f1:.4f}")
    print(f"  Latency   : {latency_total_ms:.2f} ms total, "
          f"{latency_per_cluster_ms:.3f} ms/cluster")

    # 6.3 β€” policy simulation on holdout
    policy_counts = {"auto_hold": 0, "escalate": 0, "log_only": 0}
    for i, row in enumerate(holdout):
        prob = float(probs[i])
        exposure = row["features"].get("total_amount", 0.0)
        decision = policy_decide(prob, exposure)
        policy_counts[decision] += 1

    print(f"\n--- Policy simulation on holdout ---")
    for action, count in policy_counts.items():
        print(f"  {action:12s}: {count}")

    # 6.4 β€” false-positive cost
    fp_cost = 0.0
    fp_clusters = []
    for i, row in enumerate(holdout):
        if y_pred[i] == 1 and y_true[i] == 0:
            amt = row["features"].get("total_amount", 0.0)
            fp_cost += amt
            fp_clusters.append({
                "cluster_id": row["cluster_id"],
                "total_amount": amt,
                "risk_probability": float(probs[i]),
            })

    print(f"\n--- False-positive cost ---")
    print(f"  FP clusters: {len(fp_clusters)}")
    print(f"  FP cost (total_amount): Rs.{fp_cost:.2f}")

    # 6.5 β€” latency already computed above
    print(f"\n--- Latency ---")
    print(f"  Feature extraction + inference: {latency_per_cluster_ms:.3f} ms/cluster")

    # 6.6 β€” write eval_results.json
    os.makedirs(EVAL_DIR, exist_ok=True)
    results = {
        "precision": precision,
        "recall": recall,
        "f1": f1,
        "classification_threshold": CLASSIFICATION_THRESHOLD,
        "holdout_size": len(holdout),
        "holdout_positives": int(y_true.sum()),
        "holdout_negatives": int((len(y_true) - y_true.sum())),
        "false_positive_cost_inr": fp_cost,
        "false_positive_clusters": fp_clusters,
        "latency_total_ms": latency_total_ms,
        "latency_per_cluster_ms": latency_per_cluster_ms,
        "policy_action_breakdown": policy_counts,
        "confusion_matrix": {
            "true_positive": int(((y_pred == 1) & (y_true == 1)).sum()),
            "false_positive": int(((y_pred == 1) & (y_true == 0)).sum()),
            "true_negative": int(((y_pred == 0) & (y_true == 0)).sum()),
            "false_negative": int(((y_pred == 0) & (y_true == 1)).sum()),
        },
    }

    with open(RESULTS_PATH, "w", encoding="utf-8") as f:
        json.dump(results, f, indent=2, ensure_ascii=False)
    print(f"\nEvaluation results saved to {RESULTS_PATH}")

    return results


if __name__ == "__main__":
    evaluate()