Rohit Patil
demo-day: rebuild BOUNDARY01 as genuine hard case, fix evaluate.py threshold, regenerate pipeline
f2ca89c Download eval/evaluate.py from OGrohit/ringguard: direct link, hf CLI and curl.
- Browser
- Download file 5.43 kB
-
https://huggingface.co/spaces/OGrohit/ringguard/resolve/main/eval/evaluate.py
- Command line
-
hf download hf://spaces/OGrohit/ringguard/eval/evaluate.py
-
curl -L -o evaluate.py https://huggingface.co/spaces/OGrohit/ringguard/resolve/main/eval/evaluate.py
5.43 kB
| """ | |
| evaluate.py | |
| ----------- | |
| Evaluate the trained LightGBM model on the held-out split. | |
| Outputs: | |
| eval/eval_results.json β precision, recall, F1, false-positive cost, | |
| latency, and policy-action breakdown. | |
| Usage: | |
| python eval/evaluate.py | |
| """ | |
| import json | |
| import os | |
| import sys | |
| import time | |
| import lightgbm as lgb | |
| import numpy as np | |
| from sklearn.metrics import precision_score, recall_score, f1_score | |
| BASE_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) | |
| sys.path.insert(0, BASE_DIR) | |
| SCORING_DIR = os.path.join(BASE_DIR, "scoring") | |
| EVAL_DIR = os.path.join(BASE_DIR, "eval") | |
| MODEL_PATH = os.path.join(SCORING_DIR, "model.lgb") | |
| HOLDOUT_PATH = os.path.join(SCORING_DIR, "holdout_split.json") | |
| RESULTS_PATH = os.path.join(EVAL_DIR, "eval_results.json") | |
| CLASSIFICATION_THRESHOLD = 0.5 | |
| # Policy thresholds (must match agent/policy.py exactly) | |
| POLICY_HIGH_THRESHOLD = 0.85 | |
| POLICY_MEDIUM_THRESHOLD = 0.50 | |
| POLICY_LOW_EXPOSURE = 25000.0 # updated Phase 3 v3.0 β matches agent/policy.py | |
| def policy_decide(risk_probability: float, exposure_rupees: float) -> str: | |
| """Deterministic policy gate β mirrors agent/policy.py exactly.""" | |
| if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees <= POLICY_LOW_EXPOSURE: | |
| return "auto_hold" | |
| if risk_probability >= POLICY_HIGH_THRESHOLD and exposure_rupees > POLICY_LOW_EXPOSURE: | |
| return "escalate" | |
| if POLICY_MEDIUM_THRESHOLD <= risk_probability < POLICY_HIGH_THRESHOLD: | |
| return "escalate" | |
| return "log_only" | |
| def evaluate(): | |
| print(f"Loading model from {MODEL_PATH}...") | |
| model = lgb.Booster(model_file=MODEL_PATH) | |
| print(f"Loading holdout split from {HOLDOUT_PATH}...") | |
| with open(HOLDOUT_PATH, encoding="utf-8") as f: | |
| holdout = json.load(f) | |
| print(f" {len(holdout)} held-out clusters") | |
| # 6.1 β reconstruct feature matrix | |
| from scoring.features import FEATURE_COLUMNS | |
| feature_rows = [row["features"] for row in holdout] | |
| y_true = np.array([row["label"] for row in holdout]) | |
| # 6.2 β time the feature extraction + inference pass | |
| start = time.perf_counter() | |
| X = np.array([[fr[f] for f in FEATURE_COLUMNS] for fr in feature_rows]) | |
| probs = model.predict(X) | |
| elapsed = time.perf_counter() - start | |
| y_pred = (probs >= CLASSIFICATION_THRESHOLD).astype(int) | |
| precision = float(precision_score(y_true, y_pred, zero_division=0)) | |
| recall = float(recall_score(y_true, y_pred, zero_division=0)) | |
| f1 = float(f1_score(y_true, y_pred, zero_division=0)) | |
| latency_total_ms = elapsed * 1000 | |
| latency_per_cluster_ms = latency_total_ms / len(holdout) | |
| print(f"\n--- Classification at threshold {CLASSIFICATION_THRESHOLD} ---") | |
| print(f" Precision : {precision:.4f}") | |
| print(f" Recall : {recall:.4f}") | |
| print(f" F1 : {f1:.4f}") | |
| print(f" Latency : {latency_total_ms:.2f} ms total, " | |
| f"{latency_per_cluster_ms:.3f} ms/cluster") | |
| # 6.3 β policy simulation on holdout | |
| policy_counts = {"auto_hold": 0, "escalate": 0, "log_only": 0} | |
| for i, row in enumerate(holdout): | |
| prob = float(probs[i]) | |
| exposure = row["features"].get("total_amount", 0.0) | |
| decision = policy_decide(prob, exposure) | |
| policy_counts[decision] += 1 | |
| print(f"\n--- Policy simulation on holdout ---") | |
| for action, count in policy_counts.items(): | |
| print(f" {action:12s}: {count}") | |
| # 6.4 β false-positive cost | |
| fp_cost = 0.0 | |
| fp_clusters = [] | |
| for i, row in enumerate(holdout): | |
| if y_pred[i] == 1 and y_true[i] == 0: | |
| amt = row["features"].get("total_amount", 0.0) | |
| fp_cost += amt | |
| fp_clusters.append({ | |
| "cluster_id": row["cluster_id"], | |
| "total_amount": amt, | |
| "risk_probability": float(probs[i]), | |
| }) | |
| print(f"\n--- False-positive cost ---") | |
| print(f" FP clusters: {len(fp_clusters)}") | |
| print(f" FP cost (total_amount): Rs.{fp_cost:.2f}") | |
| # 6.5 β latency already computed above | |
| print(f"\n--- Latency ---") | |
| print(f" Feature extraction + inference: {latency_per_cluster_ms:.3f} ms/cluster") | |
| # 6.6 β write eval_results.json | |
| os.makedirs(EVAL_DIR, exist_ok=True) | |
| results = { | |
| "precision": precision, | |
| "recall": recall, | |
| "f1": f1, | |
| "classification_threshold": CLASSIFICATION_THRESHOLD, | |
| "holdout_size": len(holdout), | |
| "holdout_positives": int(y_true.sum()), | |
| "holdout_negatives": int((len(y_true) - y_true.sum())), | |
| "false_positive_cost_inr": fp_cost, | |
| "false_positive_clusters": fp_clusters, | |
| "latency_total_ms": latency_total_ms, | |
| "latency_per_cluster_ms": latency_per_cluster_ms, | |
| "policy_action_breakdown": policy_counts, | |
| "confusion_matrix": { | |
| "true_positive": int(((y_pred == 1) & (y_true == 1)).sum()), | |
| "false_positive": int(((y_pred == 1) & (y_true == 0)).sum()), | |
| "true_negative": int(((y_pred == 0) & (y_true == 0)).sum()), | |
| "false_negative": int(((y_pred == 0) & (y_true == 1)).sum()), | |
| }, | |
| } | |
| with open(RESULTS_PATH, "w", encoding="utf-8") as f: | |
| json.dump(results, f, indent=2, ensure_ascii=False) | |
| print(f"\nEvaluation results saved to {RESULTS_PATH}") | |
| return results | |
| if __name__ == "__main__": | |
| evaluate() | |