""" Real ensemble construction (reviewer priority #5). The paper's title claims "Ensemble Learning" but the reported result is a single best model (LightGBM) chosen from eleven independently-trained candidates — no prediction combination happens. The reviewer explicitly asks that this either be fixed (build an actual ensemble and report its score against LightGBM) or the terminology be revised. This script loads the saved ML models (models/model_*.pkl, scaled input) and DL models (models/model_dl_*.pkl, TorchSklearnWrapper — raw input, wrapper scales internally), gets out-of-fold-style predictions via a fresh 5-fold CV using the SAME fold assignment as train_classifier.py (StratifiedKFold, random_state=42), and combines them two ways: - soft voting (unweighted mean of probabilities) - stacking (logistic regression meta-learner on the base probabilities) Both are compared against the best single model (LightGBM) on identical folds, so the comparison is apples-to-apples. Usage: python -m app.training.ensemble_model """ from __future__ import annotations import csv import sys import warnings from pathlib import Path import numpy as np sys.path.insert(0, str(Path(__file__).resolve().parents[2])) import pickle from sklearn.exceptions import ConvergenceWarning from sklearn.linear_model import LogisticRegression from sklearn.metrics import ( accuracy_score, balanced_accuracy_score, f1_score, matthews_corrcoef, roc_auc_score, roc_curve, ) from sklearn.model_selection import StratifiedKFold from sklearn.preprocessing import StandardScaler from app.training.evaluate import load_features_csv FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv") MODELS_DIR = Path(__file__).resolve().parents[2] / "models" TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables" DL_OOF_NPZ = MODELS_DIR / "dl_oof_probs.npz" # DL out-of-fold probabilities come from dump_dl_oof.py, which retrains each # of the 4 architectures per fold with the SAME StratifiedKFold(random_state # =42) split used below — so they are genuinely held-out and directly # comparable to the ML models' per-fold-refit predictions. _DL_NPZ_KEYS = { "Deep MLP": "Deep_MLP_512_256_128_64", "1D-CNN": "1D_CNN", "Residual MLP": "Residual_MLP_3_blocks", "Attention MLP": "Attention_MLP", } _ML_MODEL_FILES = { "Logistic Regression": "model_logistic_regression.pkl", "Random Forest": "model_random_forest.pkl", "Gradient Boosting": "model_gradient_boosting.pkl", "SVM (RBF)": "model_svm_rbf.pkl", "MLP Neural Network": "model_mlp_neural_network.pkl", "XGBoost": "model_xgboost.pkl", "LightGBM": "model_lightgbm.pkl", } def _load_model(filename: str): with open(MODELS_DIR / filename, "rb") as f: return pickle.load(f) def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float: fpr, tpr, thresholds = roc_curve(y_true, y_prob) return float(thresholds[np.argmax(tpr - fpr)]) def _metrics(y_true: np.ndarray, y_prob: np.ndarray) -> dict: threshold = _optimal_threshold(y_true, y_prob) y_pred = (y_prob >= threshold).astype(int) return { "roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4), "accuracy": round(float(accuracy_score(y_true, y_pred)), 4), "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4), "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4), "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4), "threshold": round(threshold, 4), } def run() -> None: X, y = load_features_csv(FEATURES_CSV) X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0) if not DL_OOF_NPZ.exists(): raise RuntimeError( f"{DL_OOF_NPZ} not found — run `python -m app.training.dump_dl_oof` first " "to generate genuinely held-out DL predictions." ) print("ML models are refit per fold below (fast). DL out-of-fold predictions") print(f"are loaded from {DL_OOF_NPZ.name} (produced by dump_dl_oof.py, which") print("retrains each DL architecture per fold on the identical fold split).\n") ml_models_raw = {name: _load_model(fname) for name, fname in _ML_MODEL_FILES.items()} all_names = list(_ML_MODEL_FILES.keys()) + list(_DL_NPZ_KEYS.keys()) dl_npz = np.load(DL_OOF_NPZ) y_npz = dl_npz["y"] # Uses the SAME StratifiedKFold(random_state=42) as train_classifier.py's # 5-fold CV and the SAME SEED=42 as train_deep_classifiers.py / dump_dl_oof.py, # so this fold assignment matches the folds the DL OOF array was computed on. cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42) fold_assignments = list(cv.split(X, y)) n = len(y) if not np.array_equal(y_npz, y): raise RuntimeError( "Label array in dl_oof_probs.npz does not match the current " "features.csv load order — DL OOF predictions would be " "misaligned with ML predictions. Re-run dump_dl_oof.py." ) oof_probs = {name: np.zeros(n) for name in all_names} for name, npz_key in _DL_NPZ_KEYS.items(): oof_probs[name] = dl_npz[npz_key] for fold_idx, (train_idx, test_idx) in enumerate(fold_assignments, start=1): print(f"Fold {fold_idx}/5 (ML models) ...") X_train, y_train = X[train_idx], y[train_idx] X_test, y_test = X[test_idx], y[test_idx] scaler = StandardScaler() X_train_scaled = scaler.fit_transform(X_train) X_test_scaled = scaler.transform(X_test) from sklearn.base import clone for name, fname in _ML_MODEL_FILES.items(): base_model = ml_models_raw[name] model = clone(base_model) with warnings.catch_warnings(): warnings.simplefilter("ignore", category=ConvergenceWarning) model.fit(X_train_scaled, y_train) oof_probs[name][test_idx] = model.predict_proba(X_test_scaled)[:, 1] print("\nPer-model OOF metrics (all models refit per fold, held-out predictions):") per_model_results = [] for name in all_names: m = _metrics(y, oof_probs[name]) m["model"] = name per_model_results.append(m) print(f" {name:25s} AUC={m['roc_auc']:.4f} F1={m['f1']:.4f}") # ── Soft voting: unweighted mean of all 11 base-model probabilities ── prob_matrix = np.column_stack([oof_probs[name] for name in all_names]) soft_vote_prob = prob_matrix.mean(axis=1) soft_vote_metrics = _metrics(y, soft_vote_prob) soft_vote_metrics["model"] = "Ensemble (soft voting, 11 models)" print(f"\n {'Ensemble (soft voting)':25s} AUC={soft_vote_metrics['roc_auc']:.4f} " f"F1={soft_vote_metrics['f1']:.4f}") # ── Stacking: logistic regression meta-learner, 5-fold on base OOF probs ── meta_cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=7) stack_prob = np.zeros(n) for train_idx, test_idx in meta_cv.split(prob_matrix, y): meta = LogisticRegression(max_iter=1000) meta.fit(prob_matrix[train_idx], y[train_idx]) stack_prob[test_idx] = meta.predict_proba(prob_matrix[test_idx])[:, 1] stack_metrics = _metrics(y, stack_prob) stack_metrics["model"] = "Ensemble (stacking, LR meta-learner)" print(f" {'Ensemble (stacking)':25s} AUC={stack_metrics['roc_auc']:.4f} " f"F1={stack_metrics['f1']:.4f}") best_single = max(per_model_results, key=lambda r: r["roc_auc"]) print("\n" + "=" * 70) print(f" Best single model: {best_single['model']} AUC={best_single['roc_auc']:.4f}") print(f" Ensemble (soft voting): AUC={soft_vote_metrics['roc_auc']:.4f} " f"(diff={soft_vote_metrics['roc_auc'] - best_single['roc_auc']:+.4f})") print(f" Ensemble (stacking): AUC={stack_metrics['roc_auc']:.4f} " f"(diff={stack_metrics['roc_auc'] - best_single['roc_auc']:+.4f})") print("=" * 70) TABLES_DIR.mkdir(parents=True, exist_ok=True) out_path = TABLES_DIR / "ensemble_comparison.csv" fieldnames = ["model", "roc_auc", "accuracy", "f1", "balanced_accuracy", "mcc", "threshold"] all_results = per_model_results + [soft_vote_metrics, stack_metrics] with open(out_path, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=fieldnames) writer.writeheader() for r in sorted(all_results, key=lambda r: -r["roc_auc"]): writer.writerow({k: r[k] for k in fieldnames}) print(f"\nOutput: {out_path}") if __name__ == "__main__": run()