File size: 9,847 Bytes
906c392
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
"""
Nested cross-validation and an independent held-out test set (reviewer
priority #4).

Two separate concerns the reviewer raised:

1. The reported hyperparameters were selected on a single stratified
   holdout, then the SAME hyperparameters were evaluated with plain 5-fold
   CV. That lets a lucky hyperparameter choice leak into the reported CV
   score. Nested CV (outer loop for evaluation, inner loop for hyperparameter
   selection, refit per outer fold) removes that leakage.

2. The reviewer explicitly warns that if the Youden J threshold is tuned on
   the same data used to report performance, "test information" has
   implicitly entered the model. This script fixes that: split
   train/val/test 70/15/15 (stratified), select the threshold on train+val
   only, then apply that FIXED threshold to the untouched test set and report
   test-set metrics.

Usage:
    python -m app.training.nested_cv
"""

from __future__ import annotations

import csv
import sys
import warnings
from pathlib import Path

import numpy as np

sys.path.insert(0, str(Path(__file__).resolve().parents[2]))

import lightgbm as lgb
from sklearn.exceptions import ConvergenceWarning
from sklearn.metrics import (
    accuracy_score,
    average_precision_score,
    balanced_accuracy_score,
    f1_score,
    matthews_corrcoef,
    precision_score,
    recall_score,
    roc_auc_score,
    roc_curve,
)
from sklearn.model_selection import StratifiedKFold, train_test_split
from sklearn.preprocessing import StandardScaler

from app.training.evaluate import load_features_csv

FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv")
TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables"

# Same search space as train_classifier.py's LightGBM candidates, used here
# for the inner-loop hyperparameter search.
_LGBM_CANDIDATES = [
    dict(n_estimators=300, max_depth=-1, learning_rate=0.05, num_leaves=31,
         subsample=0.8, colsample_bytree=0.8, min_child_samples=20,
         reg_alpha=0.1, reg_lambda=1.0),
    dict(n_estimators=500, max_depth=8, learning_rate=0.03, num_leaves=24,
         subsample=0.9, colsample_bytree=0.8, min_child_samples=30,
         reg_alpha=0.2, reg_lambda=1.2),
    dict(n_estimators=220, max_depth=6, learning_rate=0.07, num_leaves=18,
         subsample=0.75, colsample_bytree=0.75, min_child_samples=24,
         reg_alpha=0.3, reg_lambda=1.5),
]


def _fit_lgbm(params: dict, X: np.ndarray, y: np.ndarray) -> lgb.LGBMClassifier:
    model = lgb.LGBMClassifier(**params, class_weight="balanced", random_state=42, verbose=-1)
    with warnings.catch_warnings():
        warnings.simplefilter("ignore", category=ConvergenceWarning)
        model.fit(X, y)
    return model


def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float:
    fpr, tpr, thresholds = roc_curve(y_true, y_prob)
    return float(thresholds[np.argmax(tpr - fpr)])


def _full_metrics(y_true: np.ndarray, y_prob: np.ndarray, threshold: float) -> dict:
    y_pred = (y_prob >= threshold).astype(int)
    return {
        "accuracy": round(float(accuracy_score(y_true, y_pred)), 4),
        "precision": round(float(precision_score(y_true, y_pred, zero_division=0)), 4),
        "recall": round(float(recall_score(y_true, y_pred, zero_division=0)), 4),
        "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4),
        "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4),
        "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4),
        "roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4),
        "pr_auc": round(float(average_precision_score(y_true, y_prob)), 4),
    }


def nested_cv(X: np.ndarray, y: np.ndarray, outer_folds: int = 5, inner_folds: int = 3) -> list[dict]:
    """Outer loop evaluates; inner loop selects hyperparameters per outer fold."""
    outer_cv = StratifiedKFold(n_splits=outer_folds, shuffle=True, random_state=42)
    results = []

    for fold_idx, (train_idx, test_idx) in enumerate(outer_cv.split(X, y), start=1):
        X_train_outer, y_train_outer = X[train_idx], y[train_idx]
        X_test_outer, y_test_outer = X[test_idx], y[test_idx]

        scaler_outer = StandardScaler()
        X_train_outer_scaled = scaler_outer.fit_transform(X_train_outer)
        X_test_outer_scaled = scaler_outer.transform(X_test_outer)

        # Inner loop: pick the best candidate params by mean inner-fold AUC
        inner_cv = StratifiedKFold(n_splits=inner_folds, shuffle=True, random_state=fold_idx)
        best_params = None
        best_inner_auc = -1.0

        for params in _LGBM_CANDIDATES:
            inner_aucs = []
            for inner_train_idx, inner_val_idx in inner_cv.split(X_train_outer_scaled, y_train_outer):
                model = _fit_lgbm(
                    params,
                    X_train_outer_scaled[inner_train_idx],
                    y_train_outer[inner_train_idx],
                )
                y_prob_inner = model.predict_proba(X_train_outer_scaled[inner_val_idx])[:, 1]
                inner_aucs.append(roc_auc_score(y_train_outer[inner_val_idx], y_prob_inner))
            mean_inner_auc = float(np.mean(inner_aucs))
            if mean_inner_auc > best_inner_auc:
                best_inner_auc = mean_inner_auc
                best_params = params

        # Refit on the full outer-train split with the winning params
        final_model = _fit_lgbm(best_params, X_train_outer_scaled, y_train_outer)
        y_prob_train = final_model.predict_proba(X_train_outer_scaled)[:, 1]
        threshold = _optimal_threshold(y_train_outer, y_prob_train)

        y_prob_test = final_model.predict_proba(X_test_outer_scaled)[:, 1]
        metrics = _full_metrics(y_test_outer, y_prob_test, threshold)
        metrics["outer_fold"] = fold_idx
        metrics["inner_selected_n_estimators"] = best_params["n_estimators"]
        metrics["inner_selected_max_depth"] = best_params["max_depth"]
        metrics["inner_val_auc"] = round(best_inner_auc, 4)
        results.append(metrics)

        print(
            f"  Outer fold {fold_idx}: inner-selected params -> "
            f"n_estimators={best_params['n_estimators']}, max_depth={best_params['max_depth']} "
            f"| outer-test AUC={metrics['roc_auc']:.4f} F1={metrics['f1']:.4f}"
        )

    return results


def independent_test_split(X: np.ndarray, y: np.ndarray) -> dict:
    """
    70/15/15 stratified train/val/test split. Threshold is selected on
    train+val ONLY, then applied as a fixed value to the untouched test set.
    """
    X_temp, X_test, y_temp, y_test = train_test_split(
        X, y, test_size=0.15, stratify=y, random_state=42,
    )
    X_train, X_val, y_train, y_val = train_test_split(
        X_temp, y_temp, test_size=0.15 / 0.85, stratify=y_temp, random_state=42,
    )

    print(f"\n  Split sizes: train={len(y_train)} val={len(y_val)} test={len(y_test)}")

    scaler = StandardScaler()
    X_train_scaled = scaler.fit_transform(X_train)
    X_val_scaled = scaler.transform(X_val)
    X_test_scaled = scaler.transform(X_test)

    # Train on train split, select threshold on val split only
    model = _fit_lgbm(_LGBM_CANDIDATES[0], X_train_scaled, y_train)
    y_prob_val = model.predict_proba(X_val_scaled)[:, 1]
    threshold = _optimal_threshold(y_val, y_prob_val)
    print(f"  Threshold selected on VAL split only: theta* = {threshold:.4f}")

    # Refit on train+val (standard practice once threshold is frozen),
    # evaluate once on the untouched test split with the frozen threshold.
    X_trainval_scaled = np.vstack([X_train_scaled, X_val_scaled])
    y_trainval = np.concatenate([y_train, y_val])
    final_model = _fit_lgbm(_LGBM_CANDIDATES[0], X_trainval_scaled, y_trainval)

    y_prob_test = final_model.predict_proba(X_test_scaled)[:, 1]
    metrics = _full_metrics(y_test, y_prob_test, threshold)
    metrics["threshold_source"] = "train+val only (frozen before touching test)"
    metrics["n_train"] = len(y_train)
    metrics["n_val"] = len(y_val)
    metrics["n_test"] = len(y_test)
    metrics["threshold"] = round(threshold, 4)

    print(f"  Independent test-set metrics: {metrics}")
    return metrics


def run() -> None:
    X, y = load_features_csv(FEATURES_CSV)
    X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0)
    TABLES_DIR.mkdir(parents=True, exist_ok=True)

    print("=" * 70)
    print("STEP 1/2 — Nested cross-validation (outer=5, inner=3)")
    print("=" * 70)
    nested_results = nested_cv(X, y)

    nested_fieldnames = [
        "outer_fold", "inner_selected_n_estimators", "inner_selected_max_depth",
        "inner_val_auc", "accuracy", "precision", "recall", "f1",
        "balanced_accuracy", "mcc", "roc_auc", "pr_auc",
    ]
    nested_path = TABLES_DIR / "nested_cv_results.csv"
    with open(nested_path, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=nested_fieldnames)
        writer.writeheader()
        for r in nested_results:
            writer.writerow({k: r[k] for k in nested_fieldnames})

    aucs = [r["roc_auc"] for r in nested_results]
    print(f"\n  Nested CV mean AUC: {np.mean(aucs):.4f} +/- {np.std(aucs):.4f}")
    print(f"  (reference: plain 5-fold CV AUC = 0.9548)")
    print(f"  Output: {nested_path}")

    print("\n" + "=" * 70)
    print("STEP 2/2 — Independent 70/15/15 train/val/test split")
    print("=" * 70)
    independent_metrics = independent_test_split(X, y)

    independent_path = TABLES_DIR / "independent_test_results.csv"
    with open(independent_path, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=list(independent_metrics.keys()))
        writer.writeheader()
        writer.writerow(independent_metrics)
    print(f"\n  Output: {independent_path}")


if __name__ == "__main__":
    run()