""" evaluate.py ----------- Evaluate the trained LightGBM model on the held-out split. Outputs: eval/eval_results.json — precision, recall, F1, false-positive cost, latency, and policy-action breakdown. Usage: python eval/evaluate.py """ import json import os import sys import time import lightgbm as lgb import numpy as np from sklearn.metrics import precision_score, recall_score, f1_score BASE_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, BASE_DIR) SCORING_DIR = os.path.join(BASE_DIR, "scoring") EVAL_DIR = os.path.join(BASE_DIR, "eval") MODEL_PATH = os.path.join(SCORING_DIR, "model.lgb") HOLDOUT_PATH = os.path.join(SCORING_DIR, "holdout_split.json") RESULTS_PATH = os.path.join(EVAL_DIR, "eval_results.json") CLASSIFICATION_THRESHOLD = 0.5 # Policy thresholds (must match agent/policy.py exactly) POLICY_HIGH_THRESHOLD = 0.85 POLICY_MEDIUM_THRESHOLD = 0.50 POLICY_LOW_EXPOSURE = 25000.0 # updated Phase 3 v3.0 — matches agent/policy.py def policy_decide(risk_probability: float, exposure_rupees: float) -> str: """Deterministic policy gate — mirrors agent/policy.py exactly.""" if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees <= POLICY_LOW_EXPOSURE: return "auto_hold" if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees > POLICY_LOW_EXPOSURE: return "escalate" if POLICY_MEDIUM_THRESHOLD <= risk_probability < POLICY_HIGH_THRESHOLD: return "escalate" return "log_only" def evaluate(): print(f"Loading model from {MODEL_PATH}...") model = lgb.Booster(model_file=MODEL_PATH) print(f"Loading holdout split from {HOLDOUT_PATH}...") with open(HOLDOUT_PATH, encoding="utf-8") as f: holdout = json.load(f) print(f" {len(holdout)} held-out clusters") # 6.1 — reconstruct feature matrix from scoring.features import FEATURE_COLUMNS feature_rows = [row["features"] for row in holdout] y_true = np.array([row["label"] for row in holdout]) # 6.2 — time the feature extraction + inference pass start = time.perf_counter() X = np.array([[fr[f] for f in FEATURE_COLUMNS] for fr in feature_rows]) probs = model.predict(X) elapsed = time.perf_counter() - start y_pred = (probs >= CLASSIFICATION_THRESHOLD).astype(int) precision = float(precision_score(y_true, y_pred, zero_division=0)) recall = float(recall_score(y_true, y_pred, zero_division=0)) f1 = float(f1_score(y_true, y_pred, zero_division=0)) latency_total_ms = elapsed * 1000 latency_per_cluster_ms = latency_total_ms / len(holdout) print(f"\n--- Classification at threshold {CLASSIFICATION_THRESHOLD} ---") print(f" Precision : {precision:.4f}") print(f" Recall : {recall:.4f}") print(f" F1 : {f1:.4f}") print(f" Latency : {latency_total_ms:.2f} ms total, " f"{latency_per_cluster_ms:.3f} ms/cluster") # 6.3 — policy simulation on holdout policy_counts = {"auto_hold": 0, "escalate": 0, "log_only": 0} for i, row in enumerate(holdout): prob = float(probs[i]) exposure = row["features"].get("total_amount", 0.0) decision = policy_decide(prob, exposure) policy_counts[decision] += 1 print(f"\n--- Policy simulation on holdout ---") for action, count in policy_counts.items(): print(f" {action:12s}: {count}") # 6.4 — false-positive cost fp_cost = 0.0 fp_clusters = [] for i, row in enumerate(holdout): if y_pred[i] == 1 and y_true[i] == 0: amt = row["features"].get("total_amount", 0.0) fp_cost += amt fp_clusters.append({ "cluster_id": row["cluster_id"], "total_amount": amt, "risk_probability": float(probs[i]), }) print(f"\n--- False-positive cost ---") print(f" FP clusters: {len(fp_clusters)}") print(f" FP cost (total_amount): Rs.{fp_cost:.2f}") # 6.5 — latency already computed above print(f"\n--- Latency ---") print(f" Feature extraction + inference: {latency_per_cluster_ms:.3f} ms/cluster") # 6.6 — write eval_results.json os.makedirs(EVAL_DIR, exist_ok=True) results = { "precision": precision, "recall": recall, "f1": f1, "classification_threshold": CLASSIFICATION_THRESHOLD, "holdout_size": len(holdout), "holdout_positives": int(y_true.sum()), "holdout_negatives": int((len(y_true) - y_true.sum())), "false_positive_cost_inr": fp_cost, "false_positive_clusters": fp_clusters, "latency_total_ms": latency_total_ms, "latency_per_cluster_ms": latency_per_cluster_ms, "policy_action_breakdown": policy_counts, "confusion_matrix": { "true_positive": int(((y_pred == 1) & (y_true == 1)).sum()), "false_positive": int(((y_pred == 1) & (y_true == 0)).sum()), "true_negative": int(((y_pred == 0) & (y_true == 0)).sum()), "false_negative": int(((y_pred == 0) & (y_true == 1)).sum()), }, } with open(RESULTS_PATH, "w", encoding="utf-8") as f: json.dump(results, f, indent=2, ensure_ascii=False) print(f"\nEvaluation results saved to {RESULTS_PATH}") return results if __name__ == "__main__": evaluate()