crowncode-backend / app /training /ensemble_model.py
Rthur2003's picture
feat: implement automated training diagnostics, dataset bias analysis, and revision figure generation pipelines
906c392
Raw History Blame Contribute Delete
8.63 kB
"""
Real ensemble construction (reviewer priority #5).
The paper's title claims "Ensemble Learning" but the reported result is a
single best model (LightGBM) chosen from eleven independently-trained
candidates — no prediction combination happens. The reviewer explicitly asks
that this either be fixed (build an actual ensemble and report its score
against LightGBM) or the terminology be revised.
This script loads the saved ML models (models/model_*.pkl, scaled input) and
DL models (models/model_dl_*.pkl, TorchSklearnWrapper — raw input, wrapper
scales internally), gets out-of-fold-style predictions via a fresh 5-fold CV
using the SAME fold assignment as train_classifier.py (StratifiedKFold,
random_state=42), and combines them two ways:
- soft voting (unweighted mean of probabilities)
- stacking (logistic regression meta-learner on the base probabilities)
Both are compared against the best single model (LightGBM) on identical
folds, so the comparison is apples-to-apples.
Usage:
python -m app.training.ensemble_model
"""
from __future__ import annotations
import csv
import sys
import warnings
from pathlib import Path
import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
import pickle
from sklearn.exceptions import ConvergenceWarning
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import (
accuracy_score,
balanced_accuracy_score,
f1_score,
matthews_corrcoef,
roc_auc_score,
roc_curve,
)
from sklearn.model_selection import StratifiedKFold
from sklearn.preprocessing import StandardScaler
from app.training.evaluate import load_features_csv
FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv")
MODELS_DIR = Path(__file__).resolve().parents[2] / "models"
TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables"
DL_OOF_NPZ = MODELS_DIR / "dl_oof_probs.npz"
# DL out-of-fold probabilities come from dump_dl_oof.py, which retrains each
# of the 4 architectures per fold with the SAME StratifiedKFold(random_state
# =42) split used below — so they are genuinely held-out and directly
# comparable to the ML models' per-fold-refit predictions.
_DL_NPZ_KEYS = {
"Deep MLP": "Deep_MLP_512_256_128_64",
"1D-CNN": "1D_CNN",
"Residual MLP": "Residual_MLP_3_blocks",
"Attention MLP": "Attention_MLP",
}
_ML_MODEL_FILES = {
"Logistic Regression": "model_logistic_regression.pkl",
"Random Forest": "model_random_forest.pkl",
"Gradient Boosting": "model_gradient_boosting.pkl",
"SVM (RBF)": "model_svm_rbf.pkl",
"MLP Neural Network": "model_mlp_neural_network.pkl",
"XGBoost": "model_xgboost.pkl",
"LightGBM": "model_lightgbm.pkl",
}
def _load_model(filename: str):
with open(MODELS_DIR / filename, "rb") as f:
return pickle.load(f)
def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float:
fpr, tpr, thresholds = roc_curve(y_true, y_prob)
return float(thresholds[np.argmax(tpr - fpr)])
def _metrics(y_true: np.ndarray, y_prob: np.ndarray) -> dict:
threshold = _optimal_threshold(y_true, y_prob)
y_pred = (y_prob >= threshold).astype(int)
return {
"roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4),
"accuracy": round(float(accuracy_score(y_true, y_pred)), 4),
"f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4),
"balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4),
"mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4),
"threshold": round(threshold, 4),
}
def run() -> None:
X, y = load_features_csv(FEATURES_CSV)
X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0)
if not DL_OOF_NPZ.exists():
raise RuntimeError(
f"{DL_OOF_NPZ} not found — run `python -m app.training.dump_dl_oof` first "
"to generate genuinely held-out DL predictions."
)
print("ML models are refit per fold below (fast). DL out-of-fold predictions")
print(f"are loaded from {DL_OOF_NPZ.name} (produced by dump_dl_oof.py, which")
print("retrains each DL architecture per fold on the identical fold split).\n")
ml_models_raw = {name: _load_model(fname) for name, fname in _ML_MODEL_FILES.items()}
all_names = list(_ML_MODEL_FILES.keys()) + list(_DL_NPZ_KEYS.keys())
dl_npz = np.load(DL_OOF_NPZ)
y_npz = dl_npz["y"]
# Uses the SAME StratifiedKFold(random_state=42) as train_classifier.py's
# 5-fold CV and the SAME SEED=42 as train_deep_classifiers.py / dump_dl_oof.py,
# so this fold assignment matches the folds the DL OOF array was computed on.
cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)
fold_assignments = list(cv.split(X, y))
n = len(y)
if not np.array_equal(y_npz, y):
raise RuntimeError(
"Label array in dl_oof_probs.npz does not match the current "
"features.csv load order — DL OOF predictions would be "
"misaligned with ML predictions. Re-run dump_dl_oof.py."
)
oof_probs = {name: np.zeros(n) for name in all_names}
for name, npz_key in _DL_NPZ_KEYS.items():
oof_probs[name] = dl_npz[npz_key]
for fold_idx, (train_idx, test_idx) in enumerate(fold_assignments, start=1):
print(f"Fold {fold_idx}/5 (ML models) ...")
X_train, y_train = X[train_idx], y[train_idx]
X_test, y_test = X[test_idx], y[test_idx]
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)
from sklearn.base import clone
for name, fname in _ML_MODEL_FILES.items():
base_model = ml_models_raw[name]
model = clone(base_model)
with warnings.catch_warnings():
warnings.simplefilter("ignore", category=ConvergenceWarning)
model.fit(X_train_scaled, y_train)
oof_probs[name][test_idx] = model.predict_proba(X_test_scaled)[:, 1]
print("\nPer-model OOF metrics (all models refit per fold, held-out predictions):")
per_model_results = []
for name in all_names:
m = _metrics(y, oof_probs[name])
m["model"] = name
per_model_results.append(m)
print(f" {name:25s} AUC={m['roc_auc']:.4f} F1={m['f1']:.4f}")
# ── Soft voting: unweighted mean of all 11 base-model probabilities ──
prob_matrix = np.column_stack([oof_probs[name] for name in all_names])
soft_vote_prob = prob_matrix.mean(axis=1)
soft_vote_metrics = _metrics(y, soft_vote_prob)
soft_vote_metrics["model"] = "Ensemble (soft voting, 11 models)"
print(f"\n {'Ensemble (soft voting)':25s} AUC={soft_vote_metrics['roc_auc']:.4f} "
f"F1={soft_vote_metrics['f1']:.4f}")
# ── Stacking: logistic regression meta-learner, 5-fold on base OOF probs ──
meta_cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=7)
stack_prob = np.zeros(n)
for train_idx, test_idx in meta_cv.split(prob_matrix, y):
meta = LogisticRegression(max_iter=1000)
meta.fit(prob_matrix[train_idx], y[train_idx])
stack_prob[test_idx] = meta.predict_proba(prob_matrix[test_idx])[:, 1]
stack_metrics = _metrics(y, stack_prob)
stack_metrics["model"] = "Ensemble (stacking, LR meta-learner)"
print(f" {'Ensemble (stacking)':25s} AUC={stack_metrics['roc_auc']:.4f} "
f"F1={stack_metrics['f1']:.4f}")
best_single = max(per_model_results, key=lambda r: r["roc_auc"])
print("\n" + "=" * 70)
print(f" Best single model: {best_single['model']} AUC={best_single['roc_auc']:.4f}")
print(f" Ensemble (soft voting): AUC={soft_vote_metrics['roc_auc']:.4f} "
f"(diff={soft_vote_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
print(f" Ensemble (stacking): AUC={stack_metrics['roc_auc']:.4f} "
f"(diff={stack_metrics['roc_auc'] - best_single['roc_auc']:+.4f})")
print("=" * 70)
TABLES_DIR.mkdir(parents=True, exist_ok=True)
out_path = TABLES_DIR / "ensemble_comparison.csv"
fieldnames = ["model", "roc_auc", "accuracy", "f1", "balanced_accuracy", "mcc", "threshold"]
all_results = per_model_results + [soft_vote_metrics, stack_metrics]
with open(out_path, "w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(f, fieldnames=fieldnames)
writer.writeheader()
for r in sorted(all_results, key=lambda r: -r["roc_auc"]):
writer.writerow({k: r[k] for k in fieldnames})
print(f"\nOutput: {out_path}")
if __name__ == "__main__":
run()