Spaces:
Running
Running
Download app/training/nested_cv.py from Rthur2003/crowncode-backend: direct link, hf CLI and curl.
- Browser
- Download file 9.85 kB
-
https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/nested_cv.py
- Command line
-
hf download hf://spaces/Rthur2003/crowncode-backend/app/training/nested_cv.py
-
curl -L -o nested_cv.py https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/nested_cv.py
9.85 kB
| """ | |
| Nested cross-validation and an independent held-out test set (reviewer | |
| priority #4). | |
| Two separate concerns the reviewer raised: | |
| 1. The reported hyperparameters were selected on a single stratified | |
| holdout, then the SAME hyperparameters were evaluated with plain 5-fold | |
| CV. That lets a lucky hyperparameter choice leak into the reported CV | |
| score. Nested CV (outer loop for evaluation, inner loop for hyperparameter | |
| selection, refit per outer fold) removes that leakage. | |
| 2. The reviewer explicitly warns that if the Youden J threshold is tuned on | |
| the same data used to report performance, "test information" has | |
| implicitly entered the model. This script fixes that: split | |
| train/val/test 70/15/15 (stratified), select the threshold on train+val | |
| only, then apply that FIXED threshold to the untouched test set and report | |
| test-set metrics. | |
| Usage: | |
| python -m app.training.nested_cv | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import sys | |
| import warnings | |
| from pathlib import Path | |
| import numpy as np | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[2])) | |
| import lightgbm as lgb | |
| from sklearn.exceptions import ConvergenceWarning | |
| from sklearn.metrics import ( | |
| accuracy_score, | |
| average_precision_score, | |
| balanced_accuracy_score, | |
| f1_score, | |
| matthews_corrcoef, | |
| precision_score, | |
| recall_score, | |
| roc_auc_score, | |
| roc_curve, | |
| ) | |
| from sklearn.model_selection import StratifiedKFold, train_test_split | |
| from sklearn.preprocessing import StandardScaler | |
| from app.training.evaluate import load_features_csv | |
| FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv") | |
| TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables" | |
| # Same search space as train_classifier.py's LightGBM candidates, used here | |
| # for the inner-loop hyperparameter search. | |
| _LGBM_CANDIDATES = [ | |
| dict(n_estimators=300, max_depth=-1, learning_rate=0.05, num_leaves=31, | |
| subsample=0.8, colsample_bytree=0.8, min_child_samples=20, | |
| reg_alpha=0.1, reg_lambda=1.0), | |
| dict(n_estimators=500, max_depth=8, learning_rate=0.03, num_leaves=24, | |
| subsample=0.9, colsample_bytree=0.8, min_child_samples=30, | |
| reg_alpha=0.2, reg_lambda=1.2), | |
| dict(n_estimators=220, max_depth=6, learning_rate=0.07, num_leaves=18, | |
| subsample=0.75, colsample_bytree=0.75, min_child_samples=24, | |
| reg_alpha=0.3, reg_lambda=1.5), | |
| ] | |
| def _fit_lgbm(params: dict, X: np.ndarray, y: np.ndarray) -> lgb.LGBMClassifier: | |
| model = lgb.LGBMClassifier(**params, class_weight="balanced", random_state=42, verbose=-1) | |
| with warnings.catch_warnings(): | |
| warnings.simplefilter("ignore", category=ConvergenceWarning) | |
| model.fit(X, y) | |
| return model | |
| def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float: | |
| fpr, tpr, thresholds = roc_curve(y_true, y_prob) | |
| return float(thresholds[np.argmax(tpr - fpr)]) | |
| def _full_metrics(y_true: np.ndarray, y_prob: np.ndarray, threshold: float) -> dict: | |
| y_pred = (y_prob >= threshold).astype(int) | |
| return { | |
| "accuracy": round(float(accuracy_score(y_true, y_pred)), 4), | |
| "precision": round(float(precision_score(y_true, y_pred, zero_division=0)), 4), | |
| "recall": round(float(recall_score(y_true, y_pred, zero_division=0)), 4), | |
| "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4), | |
| "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4), | |
| "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4), | |
| "roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4), | |
| "pr_auc": round(float(average_precision_score(y_true, y_prob)), 4), | |
| } | |
| def nested_cv(X: np.ndarray, y: np.ndarray, outer_folds: int = 5, inner_folds: int = 3) -> list[dict]: | |
| """Outer loop evaluates; inner loop selects hyperparameters per outer fold.""" | |
| outer_cv = StratifiedKFold(n_splits=outer_folds, shuffle=True, random_state=42) | |
| results = [] | |
| for fold_idx, (train_idx, test_idx) in enumerate(outer_cv.split(X, y), start=1): | |
| X_train_outer, y_train_outer = X[train_idx], y[train_idx] | |
| X_test_outer, y_test_outer = X[test_idx], y[test_idx] | |
| scaler_outer = StandardScaler() | |
| X_train_outer_scaled = scaler_outer.fit_transform(X_train_outer) | |
| X_test_outer_scaled = scaler_outer.transform(X_test_outer) | |
| # Inner loop: pick the best candidate params by mean inner-fold AUC | |
| inner_cv = StratifiedKFold(n_splits=inner_folds, shuffle=True, random_state=fold_idx) | |
| best_params = None | |
| best_inner_auc = -1.0 | |
| for params in _LGBM_CANDIDATES: | |
| inner_aucs = [] | |
| for inner_train_idx, inner_val_idx in inner_cv.split(X_train_outer_scaled, y_train_outer): | |
| model = _fit_lgbm( | |
| params, | |
| X_train_outer_scaled[inner_train_idx], | |
| y_train_outer[inner_train_idx], | |
| ) | |
| y_prob_inner = model.predict_proba(X_train_outer_scaled[inner_val_idx])[:, 1] | |
| inner_aucs.append(roc_auc_score(y_train_outer[inner_val_idx], y_prob_inner)) | |
| mean_inner_auc = float(np.mean(inner_aucs)) | |
| if mean_inner_auc > best_inner_auc: | |
| best_inner_auc = mean_inner_auc | |
| best_params = params | |
| # Refit on the full outer-train split with the winning params | |
| final_model = _fit_lgbm(best_params, X_train_outer_scaled, y_train_outer) | |
| y_prob_train = final_model.predict_proba(X_train_outer_scaled)[:, 1] | |
| threshold = _optimal_threshold(y_train_outer, y_prob_train) | |
| y_prob_test = final_model.predict_proba(X_test_outer_scaled)[:, 1] | |
| metrics = _full_metrics(y_test_outer, y_prob_test, threshold) | |
| metrics["outer_fold"] = fold_idx | |
| metrics["inner_selected_n_estimators"] = best_params["n_estimators"] | |
| metrics["inner_selected_max_depth"] = best_params["max_depth"] | |
| metrics["inner_val_auc"] = round(best_inner_auc, 4) | |
| results.append(metrics) | |
| print( | |
| f" Outer fold {fold_idx}: inner-selected params -> " | |
| f"n_estimators={best_params['n_estimators']}, max_depth={best_params['max_depth']} " | |
| f"| outer-test AUC={metrics['roc_auc']:.4f} F1={metrics['f1']:.4f}" | |
| ) | |
| return results | |
| def independent_test_split(X: np.ndarray, y: np.ndarray) -> dict: | |
| """ | |
| 70/15/15 stratified train/val/test split. Threshold is selected on | |
| train+val ONLY, then applied as a fixed value to the untouched test set. | |
| """ | |
| X_temp, X_test, y_temp, y_test = train_test_split( | |
| X, y, test_size=0.15, stratify=y, random_state=42, | |
| ) | |
| X_train, X_val, y_train, y_val = train_test_split( | |
| X_temp, y_temp, test_size=0.15 / 0.85, stratify=y_temp, random_state=42, | |
| ) | |
| print(f"\n Split sizes: train={len(y_train)} val={len(y_val)} test={len(y_test)}") | |
| scaler = StandardScaler() | |
| X_train_scaled = scaler.fit_transform(X_train) | |
| X_val_scaled = scaler.transform(X_val) | |
| X_test_scaled = scaler.transform(X_test) | |
| # Train on train split, select threshold on val split only | |
| model = _fit_lgbm(_LGBM_CANDIDATES[0], X_train_scaled, y_train) | |
| y_prob_val = model.predict_proba(X_val_scaled)[:, 1] | |
| threshold = _optimal_threshold(y_val, y_prob_val) | |
| print(f" Threshold selected on VAL split only: theta* = {threshold:.4f}") | |
| # Refit on train+val (standard practice once threshold is frozen), | |
| # evaluate once on the untouched test split with the frozen threshold. | |
| X_trainval_scaled = np.vstack([X_train_scaled, X_val_scaled]) | |
| y_trainval = np.concatenate([y_train, y_val]) | |
| final_model = _fit_lgbm(_LGBM_CANDIDATES[0], X_trainval_scaled, y_trainval) | |
| y_prob_test = final_model.predict_proba(X_test_scaled)[:, 1] | |
| metrics = _full_metrics(y_test, y_prob_test, threshold) | |
| metrics["threshold_source"] = "train+val only (frozen before touching test)" | |
| metrics["n_train"] = len(y_train) | |
| metrics["n_val"] = len(y_val) | |
| metrics["n_test"] = len(y_test) | |
| metrics["threshold"] = round(threshold, 4) | |
| print(f" Independent test-set metrics: {metrics}") | |
| return metrics | |
| def run() -> None: | |
| X, y = load_features_csv(FEATURES_CSV) | |
| X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0) | |
| TABLES_DIR.mkdir(parents=True, exist_ok=True) | |
| print("=" * 70) | |
| print("STEP 1/2 — Nested cross-validation (outer=5, inner=3)") | |
| print("=" * 70) | |
| nested_results = nested_cv(X, y) | |
| nested_fieldnames = [ | |
| "outer_fold", "inner_selected_n_estimators", "inner_selected_max_depth", | |
| "inner_val_auc", "accuracy", "precision", "recall", "f1", | |
| "balanced_accuracy", "mcc", "roc_auc", "pr_auc", | |
| ] | |
| nested_path = TABLES_DIR / "nested_cv_results.csv" | |
| with open(nested_path, "w", newline="", encoding="utf-8") as f: | |
| writer = csv.DictWriter(f, fieldnames=nested_fieldnames) | |
| writer.writeheader() | |
| for r in nested_results: | |
| writer.writerow({k: r[k] for k in nested_fieldnames}) | |
| aucs = [r["roc_auc"] for r in nested_results] | |
| print(f"\n Nested CV mean AUC: {np.mean(aucs):.4f} +/- {np.std(aucs):.4f}") | |
| print(f" (reference: plain 5-fold CV AUC = 0.9548)") | |
| print(f" Output: {nested_path}") | |
| print("\n" + "=" * 70) | |
| print("STEP 2/2 — Independent 70/15/15 train/val/test split") | |
| print("=" * 70) | |
| independent_metrics = independent_test_split(X, y) | |
| independent_path = TABLES_DIR / "independent_test_results.csv" | |
| with open(independent_path, "w", newline="", encoding="utf-8") as f: | |
| writer = csv.DictWriter(f, fieldnames=list(independent_metrics.keys())) | |
| writer.writeheader() | |
| writer.writerow(independent_metrics) | |
| print(f"\n Output: {independent_path}") | |
| if __name__ == "__main__": | |
| run() | |