Spaces:
Sleeping
Sleeping
Download app/training/ensemble_model.py from Rthur2003/crowncode-backend: direct link, hf CLI and curl.
- Browser
- Download file 8.63 kB
-
https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/ensemble_model.py
- Command line
-
hf download hf://spaces/Rthur2003/crowncode-backend/app/training/ensemble_model.py
-
curl -L -o ensemble_model.py https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/training/ensemble_model.py
8.63 kB
| """ | |
| Real ensemble construction (reviewer priority #5). | |
| The paper's title claims "Ensemble Learning" but the reported result is a | |
| single best model (LightGBM) chosen from eleven independently-trained | |
| candidates — no prediction combination happens. The reviewer explicitly asks | |
| that this either be fixed (build an actual ensemble and report its score | |
| against LightGBM) or the terminology be revised. | |
| This script loads the saved ML models (models/model_*.pkl, scaled input) and | |
| DL models (models/model_dl_*.pkl, TorchSklearnWrapper — raw input, wrapper | |
| scales internally), gets out-of-fold-style predictions via a fresh 5-fold CV | |
| using the SAME fold assignment as train_classifier.py (StratifiedKFold, | |
| random_state=42), and combines them two ways: | |
| - soft voting (unweighted mean of probabilities) | |
| - stacking (logistic regression meta-learner on the base probabilities) | |
| Both are compared against the best single model (LightGBM) on identical | |
| folds, so the comparison is apples-to-apples. | |
| Usage: | |
| python -m app.training.ensemble_model | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import sys | |
| import warnings | |
| from pathlib import Path | |
| import numpy as np | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[2])) | |
| import pickle | |
| from sklearn.exceptions import ConvergenceWarning | |
| from sklearn.linear_model import LogisticRegression | |
| from sklearn.metrics import ( | |
| accuracy_score, | |
| balanced_accuracy_score, | |
| f1_score, | |
| matthews_corrcoef, | |
| roc_auc_score, | |
| roc_curve, | |
| ) | |
| from sklearn.model_selection import StratifiedKFold | |
| from sklearn.preprocessing import StandardScaler | |
| from app.training.evaluate import load_features_csv | |
| FEATURES_CSV = Path("D:/CrownCode/DataSet/features.csv") | |
| MODELS_DIR = Path(__file__).resolve().parents[2] / "models" | |
| TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables" | |
| DL_OOF_NPZ = MODELS_DIR / "dl_oof_probs.npz" | |
| # DL out-of-fold probabilities come from dump_dl_oof.py, which retrains each | |
| # of the 4 architectures per fold with the SAME StratifiedKFold(random_state | |
| # =42) split used below — so they are genuinely held-out and directly | |
| # comparable to the ML models' per-fold-refit predictions. | |
| _DL_NPZ_KEYS = { | |
| "Deep MLP": "Deep_MLP_512_256_128_64", | |
| "1D-CNN": "1D_CNN", | |
| "Residual MLP": "Residual_MLP_3_blocks", | |
| "Attention MLP": "Attention_MLP", | |
| } | |
| _ML_MODEL_FILES = { | |
| "Logistic Regression": "model_logistic_regression.pkl", | |
| "Random Forest": "model_random_forest.pkl", | |
| "Gradient Boosting": "model_gradient_boosting.pkl", | |
| "SVM (RBF)": "model_svm_rbf.pkl", | |
| "MLP Neural Network": "model_mlp_neural_network.pkl", | |
| "XGBoost": "model_xgboost.pkl", | |
| "LightGBM": "model_lightgbm.pkl", | |
| } | |
| def _load_model(filename: str): | |
| with open(MODELS_DIR / filename, "rb") as f: | |
| return pickle.load(f) | |
| def _optimal_threshold(y_true: np.ndarray, y_prob: np.ndarray) -> float: | |
| fpr, tpr, thresholds = roc_curve(y_true, y_prob) | |
| return float(thresholds[np.argmax(tpr - fpr)]) | |
| def _metrics(y_true: np.ndarray, y_prob: np.ndarray) -> dict: | |
| threshold = _optimal_threshold(y_true, y_prob) | |
| y_pred = (y_prob >= threshold).astype(int) | |
| return { | |
| "roc_auc": round(float(roc_auc_score(y_true, y_prob)), 4), | |
| "accuracy": round(float(accuracy_score(y_true, y_pred)), 4), | |
| "f1": round(float(f1_score(y_true, y_pred, zero_division=0)), 4), | |
| "balanced_accuracy": round(float(balanced_accuracy_score(y_true, y_pred)), 4), | |
| "mcc": round(float(matthews_corrcoef(y_true, y_pred)), 4), | |
| "threshold": round(threshold, 4), | |
| } | |
| def run() -> None: | |
| X, y = load_features_csv(FEATURES_CSV) | |
| X = np.nan_to_num(X, nan=0.0, posinf=1.0, neginf=-1.0) | |
| if not DL_OOF_NPZ.exists(): | |
| raise RuntimeError( | |
| f"{DL_OOF_NPZ} not found — run `python -m app.training.dump_dl_oof` first " | |
| "to generate genuinely held-out DL predictions." | |
| ) | |
| print("ML models are refit per fold below (fast). DL out-of-fold predictions") | |
| print(f"are loaded from {DL_OOF_NPZ.name} (produced by dump_dl_oof.py, which") | |
| print("retrains each DL architecture per fold on the identical fold split).\n") | |
| ml_models_raw = {name: _load_model(fname) for name, fname in _ML_MODEL_FILES.items()} | |
| all_names = list(_ML_MODEL_FILES.keys()) + list(_DL_NPZ_KEYS.keys()) | |
| dl_npz = np.load(DL_OOF_NPZ) | |
| y_npz = dl_npz["y"] | |
| # Uses the SAME StratifiedKFold(random_state=42) as train_classifier.py's | |
| # 5-fold CV and the SAME SEED=42 as train_deep_classifiers.py / dump_dl_oof.py, | |
| # so this fold assignment matches the folds the DL OOF array was computed on. | |
| cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42) | |
| fold_assignments = list(cv.split(X, y)) | |
| n = len(y) | |
| if not np.array_equal(y_npz, y): | |
| raise RuntimeError( | |
| "Label array in dl_oof_probs.npz does not match the current " | |
| "features.csv load order — DL OOF predictions would be " | |
| "misaligned with ML predictions. Re-run dump_dl_oof.py." | |
| ) | |
| oof_probs = {name: np.zeros(n) for name in all_names} | |
| for name, npz_key in _DL_NPZ_KEYS.items(): | |
| oof_probs[name] = dl_npz[npz_key] | |
| for fold_idx, (train_idx, test_idx) in enumerate(fold_assignments, start=1): | |
| print(f"Fold {fold_idx}/5 (ML models) ...") | |
| X_train, y_train = X[train_idx], y[train_idx] | |
| X_test, y_test = X[test_idx], y[test_idx] | |
| scaler = StandardScaler() | |
| X_train_scaled = scaler.fit_transform(X_train) | |
| X_test_scaled = scaler.transform(X_test) | |
| from sklearn.base import clone | |
| for name, fname in _ML_MODEL_FILES.items(): | |
| base_model = ml_models_raw[name] | |
| model = clone(base_model) | |
| with warnings.catch_warnings(): | |
| warnings.simplefilter("ignore", category=ConvergenceWarning) | |
| model.fit(X_train_scaled, y_train) | |
| oof_probs[name][test_idx] = model.predict_proba(X_test_scaled)[:, 1] | |
| print("\nPer-model OOF metrics (all models refit per fold, held-out predictions):") | |
| per_model_results = [] | |
| for name in all_names: | |
| m = _metrics(y, oof_probs[name]) | |
| m["model"] = name | |
| per_model_results.append(m) | |
| print(f" {name:25s} AUC={m['roc_auc']:.4f} F1={m['f1']:.4f}") | |
| # ── Soft voting: unweighted mean of all 11 base-model probabilities ── | |
| prob_matrix = np.column_stack([oof_probs[name] for name in all_names]) | |
| soft_vote_prob = prob_matrix.mean(axis=1) | |
| soft_vote_metrics = _metrics(y, soft_vote_prob) | |
| soft_vote_metrics["model"] = "Ensemble (soft voting, 11 models)" | |
| print(f"\n {'Ensemble (soft voting)':25s} AUC={soft_vote_metrics['roc_auc']:.4f} " | |
| f"F1={soft_vote_metrics['f1']:.4f}") | |
| # ── Stacking: logistic regression meta-learner, 5-fold on base OOF probs ── | |
| meta_cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=7) | |
| stack_prob = np.zeros(n) | |
| for train_idx, test_idx in meta_cv.split(prob_matrix, y): | |
| meta = LogisticRegression(max_iter=1000) | |
| meta.fit(prob_matrix[train_idx], y[train_idx]) | |
| stack_prob[test_idx] = meta.predict_proba(prob_matrix[test_idx])[:, 1] | |
| stack_metrics = _metrics(y, stack_prob) | |
| stack_metrics["model"] = "Ensemble (stacking, LR meta-learner)" | |
| print(f" {'Ensemble (stacking)':25s} AUC={stack_metrics['roc_auc']:.4f} " | |
| f"F1={stack_metrics['f1']:.4f}") | |
| best_single = max(per_model_results, key=lambda r: r["roc_auc"]) | |
| print("\n" + "=" * 70) | |
| print(f" Best single model: {best_single['model']} AUC={best_single['roc_auc']:.4f}") | |
| print(f" Ensemble (soft voting): AUC={soft_vote_metrics['roc_auc']:.4f} " | |
| f"(diff={soft_vote_metrics['roc_auc'] - best_single['roc_auc']:+.4f})") | |
| print(f" Ensemble (stacking): AUC={stack_metrics['roc_auc']:.4f} " | |
| f"(diff={stack_metrics['roc_auc'] - best_single['roc_auc']:+.4f})") | |
| print("=" * 70) | |
| TABLES_DIR.mkdir(parents=True, exist_ok=True) | |
| out_path = TABLES_DIR / "ensemble_comparison.csv" | |
| fieldnames = ["model", "roc_auc", "accuracy", "f1", "balanced_accuracy", "mcc", "threshold"] | |
| all_results = per_model_results + [soft_vote_metrics, stack_metrics] | |
| with open(out_path, "w", newline="", encoding="utf-8") as f: | |
| writer = csv.DictWriter(f, fieldnames=fieldnames) | |
| writer.writeheader() | |
| for r in sorted(all_results, key=lambda r: -r["roc_auc"]): | |
| writer.writerow({k: r[k] for k in fieldnames}) | |
| print(f"\nOutput: {out_path}") | |
| if __name__ == "__main__": | |
| run() | |