Spaces:
Sleeping
Sleeping
File size: 8,629 Bytes
906c392 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 | """
Real ensemble construction (reviewer priority #5).
The paper's title claims "Ensemble Learning" but the reported result is a
single best model (LightGBM) chosen from eleven independently-trained
candidates — no prediction combination happens. The reviewer explicitly asks
that this either be fixed (build an actual ensemble and report its score
against LightGBM) or the terminology be revised.
This script loads the saved ML models (models/model_*.pkl, scaled input) and
DL models (models/model_dl_*.pkl, TorchSklearnWrapper — raw input, wrapper
scales internally), gets out-of-fold-style predictions via a fresh 5-fold CV
using the SAME fold assignment as train_classifier.py (StratifiedKFold,
random_state=42), and combines them two ways:
- soft voting (unweighted mean of probabilities)
- stacking (logistic regression meta-learner on the base probabilities)
Both are compared against the best single model (LightGBM) on identical
folds, so the comparison is apples-to-apples.
Usage:
python -m app.training.ensemble_model
"""
from __future__ import annotations
import csv
import sys
import warnings
from pathlib import Path
import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
import pickle
from sklearn.exceptions import ConvergenceWarning
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import (
accuracy_score,
balanced_accuracy_score,
f1_score,
matthews_corrcoef,
roc_auc_score,
roc_curve,
)
from sklearn.model_selection import StratifiedKFold
from sklearn.preprocessing import StandardScaler
from app.training.evaluate import load_features_csv
FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv")
MODELS_DIR = Path(__file__).resolve().parents[2] / "models"
TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables"
DL_OOF_NPZ = MODELS_DIR / "dl_oof_probs.npz"
# DL out-of-fold probabilities come from dump_dl_oof.py, which retrains each
# of the 4 architectures per fold with the SAME StratifiedKFold(random_state
# =42) split used below — so they are genuinely held-out and directly
# comparable to the ML models' per-fold-refit predictions.
_DL_NPZ_KEYS = {
"Deep MLP": "Deep_MLP_512_256_128_64",
"1D-CNN": "1D_CNN",
"Residual MLP": "Residual_MLP_3_blocks",
"Attention MLP": "Attention_MLP",
}
_ML_MODEL_FILES = {
"Logistic Regression": "model_logistic_regression.pkl",
"Random Forest": "model_random_forest.pkl",
"Gradient Boosting": "model_gradient_boosting.pkl",
"SVM (RBF)": "model_svm_rbf.pkl",
"MLP Neural Network": "model_mlp_neural_network.pkl",
"XGBoost": "model_xgboost.pkl",
"LightGBM": "model_lightgbm.pkl",
}
def _load_model(filename: str):
with open(MODELS_DIR / filename, "rb") as f:
return pickle.load(f)
def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float:
fpr, tpr, thresholds = roc_curve(y_true, y_prob)
return float(thresholds[np.argmax(tpr - fpr)])
def _metrics(y_true: np.ndarray, y_prob: np.ndarray) -> dict:
threshold = _optimal_threshold(y_true, y_prob)
y_pred = (y_prob >= threshold).astype(int)
return {
"roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4),
"accuracy": round(float(accuracy_score(y_true, y_pred)), 4),
"f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4),
"balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4),
"mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4),
"threshold": round(threshold, 4),
}
def run() -> None:
X, y = load_features_csv(FEATURES_CSV)
X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0)
if not DL_OOF_NPZ.exists():
raise RuntimeError(
f"{DL_OOF_NPZ} not found — run `python -m app.training.dump_dl_oof` first "
"to generate genuinely held-out DL predictions."
)
print("ML models are refit per fold below (fast). DL out-of-fold predictions")
print(f"are loaded from {DL_OOF_NPZ.name} (produced by dump_dl_oof.py, which")
print("retrains each DL architecture per fold on the identical fold split).\n")
ml_models_raw = {name: _load_model(fname) for name, fname in _ML_MODEL_FILES.items()}
all_names = list(_ML_MODEL_FILES.keys()) + list(_DL_NPZ_KEYS.keys())
dl_npz = np.load(DL_OOF_NPZ)
y_npz = dl_npz["y"]
# Uses the SAME StratifiedKFold(random_state=42) as train_classifier.py's
# 5-fold CV and the SAME SEED=42 as train_deep_classifiers.py / dump_dl_oof.py,
# so this fold assignment matches the folds the DL OOF array was computed on.
cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)
fold_assignments = list(cv.split(X, y))
n = len(y)
if not np.array_equal(y_npz, y):
raise RuntimeError(
"Label array in dl_oof_probs.npz does not match the current "
"features.csv load order — DL OOF predictions would be "
"misaligned with ML predictions. Re-run dump_dl_oof.py."
)
oof_probs = {name: np.zeros(n) for name in all_names}
for name, npz_key in _DL_NPZ_KEYS.items():
oof_probs[name] = dl_npz[npz_key]
for fold_idx, (train_idx, test_idx) in enumerate(fold_assignments, start=1):
print(f"Fold {fold_idx}/5 (ML models) ...")
X_train, y_train = X[train_idx], y[train_idx]
X_test, y_test = X[test_idx], y[test_idx]
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)
from sklearn.base import clone
for name, fname in _ML_MODEL_FILES.items():
base_model = ml_models_raw[name]
model = clone(base_model)
with warnings.catch_warnings():
warnings.simplefilter("ignore", category=ConvergenceWarning)
model.fit(X_train_scaled, y_train)
oof_probs[name][test_idx] = model.predict_proba(X_test_scaled)[:, 1]
print("\nPer-model OOF metrics (all models refit per fold, held-out predictions):")
per_model_results = []
for name in all_names:
m = _metrics(y, oof_probs[name])
m["model"] = name
per_model_results.append(m)
print(f" {name:25s} AUC={m['roc_auc']:.4f} F1={m['f1']:.4f}")
# ── Soft voting: unweighted mean of all 11 base-model probabilities ──
prob_matrix = np.column_stack([oof_probs[name] for name in all_names])
soft_vote_prob = prob_matrix.mean(axis=1)
soft_vote_metrics = _metrics(y, soft_vote_prob)
soft_vote_metrics["model"] = "Ensemble (soft voting, 11 models)"
print(f"\n {'Ensemble (soft voting)':25s} AUC={soft_vote_metrics['roc_auc']:.4f} "
f"F1={soft_vote_metrics['f1']:.4f}")
# ── Stacking: logistic regression meta-learner, 5-fold on base OOF probs ──
meta_cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=7)
stack_prob = np.zeros(n)
for train_idx, test_idx in meta_cv.split(prob_matrix, y):
meta = LogisticRegression(max_iter=1000)
meta.fit(prob_matrix[train_idx], y[train_idx])
stack_prob[test_idx] = meta.predict_proba(prob_matrix[test_idx])[:, 1]
stack_metrics = _metrics(y, stack_prob)
stack_metrics["model"] = "Ensemble (stacking, LR meta-learner)"
print(f" {'Ensemble (stacking)':25s} AUC={stack_metrics['roc_auc']:.4f} "
f"F1={stack_metrics['f1']:.4f}")
best_single = max(per_model_results, key=lambda r: r["roc_auc"])
print("\n" + "=" * 70)
print(f" Best single model: {best_single['model']} AUC={best_single['roc_auc']:.4f}")
print(f" Ensemble (soft voting): AUC={soft_vote_metrics['roc_auc']:.4f} "
f"(diff={soft_vote_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
print(f" Ensemble (stacking): AUC={stack_metrics['roc_auc']:.4f} "
f"(diff={stack_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
print("=" * 70)
TABLES_DIR.mkdir(parents=True, exist_ok=True)
out_path = TABLES_DIR / "ensemble_comparison.csv"
fieldnames = ["model", "roc_auc", "accuracy", "f1", "balanced_accuracy", "mcc", "threshold"]
all_results = per_model_results + [soft_vote_metrics, stack_metrics]
with open(out_path, "w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(f, fieldnames=fieldnames)
writer.writeheader()
for r in sorted(all_results, key=lambda r: -r["roc_auc"]):
writer.writerow({k: r[k] for k in fieldnames})
print(f"\nOutput: {out_path}")
if __name__ == "__main__":
run()
|