File size: 8,629 Bytes
906c392
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
"""
Real ensemble construction (reviewer priority #5).

The paper's title claims "Ensemble Learning" but the reported result is a
single best model (LightGBM) chosen from eleven independently-trained
candidates — no prediction combination happens. The reviewer explicitly asks
that this either be fixed (build an actual ensemble and report its score
against LightGBM) or the terminology be revised.

This script loads the saved ML models (models/model_*.pkl, scaled input) and
DL models (models/model_dl_*.pkl, TorchSklearnWrapper — raw input, wrapper
scales internally), gets out-of-fold-style predictions via a fresh 5-fold CV
using the SAME fold assignment as train_classifier.py (StratifiedKFold,
random_state=42), and combines them two ways:
  - soft voting (unweighted mean of probabilities)
  - stacking (logistic regression meta-learner on the base probabilities)

Both are compared against the best single model (LightGBM) on identical
folds, so the comparison is apples-to-apples.

Usage:
    python -m app.training.ensemble_model
"""

from __future__ import annotations

import csv
import sys
import warnings
from pathlib import Path

import numpy as np

sys.path.insert(0, str(Path(__file__).resolve().parents[2]))

import pickle
from sklearn.exceptions import ConvergenceWarning
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import (
    accuracy_score,
    balanced_accuracy_score,
    f1_score,
    matthews_corrcoef,
    roc_auc_score,
    roc_curve,
)
from sklearn.model_selection import StratifiedKFold
from sklearn.preprocessing import StandardScaler

from app.training.evaluate import load_features_csv

FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv")
MODELS_DIR = Path(__file__).resolve().parents[2] / "models"
TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables"
DL_OOF_NPZ = MODELS_DIR / "dl_oof_probs.npz"

# DL out-of-fold probabilities come from dump_dl_oof.py, which retrains each
# of the 4 architectures per fold with the SAME StratifiedKFold(random_state
# =42) split used below — so they are genuinely held-out and directly
# comparable to the ML models' per-fold-refit predictions.
_DL_NPZ_KEYS = {
    "Deep MLP": "Deep_MLP_512_256_128_64",
    "1D-CNN": "1D_CNN",
    "Residual MLP": "Residual_MLP_3_blocks",
    "Attention MLP": "Attention_MLP",
}
_ML_MODEL_FILES = {
    "Logistic Regression": "model_logistic_regression.pkl",
    "Random Forest": "model_random_forest.pkl",
    "Gradient Boosting": "model_gradient_boosting.pkl",
    "SVM (RBF)": "model_svm_rbf.pkl",
    "MLP Neural Network": "model_mlp_neural_network.pkl",
    "XGBoost": "model_xgboost.pkl",
    "LightGBM": "model_lightgbm.pkl",
}


def _load_model(filename: str):
    with open(MODELS_DIR / filename, "rb") as f:
        return pickle.load(f)


def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float:
    fpr, tpr, thresholds = roc_curve(y_true, y_prob)
    return float(thresholds[np.argmax(tpr - fpr)])


def _metrics(y_true: np.ndarray, y_prob: np.ndarray) -> dict:
    threshold = _optimal_threshold(y_true, y_prob)
    y_pred = (y_prob >= threshold).astype(int)
    return {
        "roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4),
        "accuracy": round(float(accuracy_score(y_true, y_pred)), 4),
        "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4),
        "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4),
        "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4),
        "threshold": round(threshold, 4),
    }


def run() -> None:
    X, y = load_features_csv(FEATURES_CSV)
    X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0)

    if not DL_OOF_NPZ.exists():
        raise RuntimeError(
            f"{DL_OOF_NPZ} not found — run `python -m app.training.dump_dl_oof` first "
            "to generate genuinely held-out DL predictions."
        )

    print("ML models are refit per fold below (fast). DL out-of-fold predictions")
    print(f"are loaded from {DL_OOF_NPZ.name} (produced by dump_dl_oof.py, which")
    print("retrains each DL architecture per fold on the identical fold split).\n")

    ml_models_raw = {name: _load_model(fname) for name, fname in _ML_MODEL_FILES.items()}
    all_names = list(_ML_MODEL_FILES.keys()) + list(_DL_NPZ_KEYS.keys())

    dl_npz = np.load(DL_OOF_NPZ)
    y_npz = dl_npz["y"]

    # Uses the SAME StratifiedKFold(random_state=42) as train_classifier.py's
    # 5-fold CV and the SAME SEED=42 as train_deep_classifiers.py / dump_dl_oof.py,
    # so this fold assignment matches the folds the DL OOF array was computed on.
    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)
    fold_assignments = list(cv.split(X, y))

    n = len(y)
    if not np.array_equal(y_npz, y):
        raise RuntimeError(
            "Label array in dl_oof_probs.npz does not match the current "
            "features.csv load order — DL OOF predictions would be "
            "misaligned with ML predictions. Re-run dump_dl_oof.py."
        )

    oof_probs = {name: np.zeros(n) for name in all_names}
    for name, npz_key in _DL_NPZ_KEYS.items():
        oof_probs[name] = dl_npz[npz_key]

    for fold_idx, (train_idx, test_idx) in enumerate(fold_assignments, start=1):
        print(f"Fold {fold_idx}/5 (ML models) ...")
        X_train, y_train = X[train_idx], y[train_idx]
        X_test, y_test = X[test_idx], y[test_idx]

        scaler = StandardScaler()
        X_train_scaled = scaler.fit_transform(X_train)
        X_test_scaled = scaler.transform(X_test)

        from sklearn.base import clone
        for name, fname in _ML_MODEL_FILES.items():
            base_model = ml_models_raw[name]
            model = clone(base_model)
            with warnings.catch_warnings():
                warnings.simplefilter("ignore", category=ConvergenceWarning)
                model.fit(X_train_scaled, y_train)
            oof_probs[name][test_idx] = model.predict_proba(X_test_scaled)[:, 1]

    print("\nPer-model OOF metrics (all models refit per fold, held-out predictions):")
    per_model_results = []
    for name in all_names:
        m = _metrics(y, oof_probs[name])
        m["model"] = name
        per_model_results.append(m)
        print(f"  {name:25s} AUC={m['roc_auc']:.4f} F1={m['f1']:.4f}")

    # ── Soft voting: unweighted mean of all 11 base-model probabilities ──
    prob_matrix = np.column_stack([oof_probs[name] for name in all_names])
    soft_vote_prob = prob_matrix.mean(axis=1)
    soft_vote_metrics = _metrics(y, soft_vote_prob)
    soft_vote_metrics["model"] = "Ensemble (soft voting, 11 models)"
    print(f"\n  {'Ensemble (soft voting)':25s} AUC={soft_vote_metrics['roc_auc']:.4f} "
          f"F1={soft_vote_metrics['f1']:.4f}")

    # ── Stacking: logistic regression meta-learner, 5-fold on base OOF probs ──
    meta_cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=7)
    stack_prob = np.zeros(n)
    for train_idx, test_idx in meta_cv.split(prob_matrix, y):
        meta = LogisticRegression(max_iter=1000)
        meta.fit(prob_matrix[train_idx], y[train_idx])
        stack_prob[test_idx] = meta.predict_proba(prob_matrix[test_idx])[:, 1]
    stack_metrics = _metrics(y, stack_prob)
    stack_metrics["model"] = "Ensemble (stacking, LR meta-learner)"
    print(f"  {'Ensemble (stacking)':25s} AUC={stack_metrics['roc_auc']:.4f} "
          f"F1={stack_metrics['f1']:.4f}")

    best_single = max(per_model_results, key=lambda r: r["roc_auc"])
    print("\n" + "=" * 70)
    print(f"  Best single model:      {best_single['model']} AUC={best_single['roc_auc']:.4f}")
    print(f"  Ensemble (soft voting): AUC={soft_vote_metrics['roc_auc']:.4f} "
          f"(diff={soft_vote_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
    print(f"  Ensemble (stacking):    AUC={stack_metrics['roc_auc']:.4f} "
          f"(diff={stack_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
    print("=" * 70)

    TABLES_DIR.mkdir(parents=True, exist_ok=True)
    out_path = TABLES_DIR / "ensemble_comparison.csv"
    fieldnames = ["model", "roc_auc", "accuracy", "f1", "balanced_accuracy", "mcc", "threshold"]
    all_results = per_model_results + [soft_vote_metrics, stack_metrics]
    with open(out_path, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=fieldnames)
        writer.writeheader()
        for r in sorted(all_results, key=lambda r: -r["roc_auc"]):
            writer.writerow({k: r[k] for k in fieldnames})

    print(f"\nOutput: {out_path}")


if __name__ == "__main__":
    run()