File size: 10,742 Bytes
8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f cbba70a 8b37c3f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 | """Leakage-aware grouped evaluation for eight classical and two ensembles."""
from __future__ import annotations
from collections.abc import Callable
import numpy as np
import pandas as pd
from sklearn.base import RegressorMixin
from sklearn.compose import TransformedTargetRegressor
from sklearn.ensemble import ExtraTreesRegressor, GradientBoostingRegressor, RandomForestRegressor
from sklearn.impute import SimpleImputer
from sklearn.linear_model import Ridge
from sklearn.neighbors import KNeighborsRegressor
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.svm import SVR
from sklearn.model_selection import GroupKFold
from src.evaluation.metrics import regression_metrics
from src.evaluation.protocol import grouped_train_val_test_folds
from src.utils.config import FEATURE_COLS_V3
def _tabular_pipeline(model: RegressorMixin, *, scale: bool = False) -> Pipeline:
steps: list[tuple[str, object]] = [
("imputer", SimpleImputer(strategy="median", keep_empty_features=True))
]
if scale:
steps.append(("scaler", StandardScaler()))
steps.append(("model", model))
return Pipeline(steps)
def classical_factories(random_state: int = 42) -> dict[str, Callable[[], RegressorMixin]]:
"""Return fresh estimator factories with deterministic CPU settings."""
from lightgbm import LGBMRegressor
from xgboost import XGBRegressor
return {
"extra_trees": lambda: _tabular_pipeline(ExtraTreesRegressor(
n_estimators=300, min_samples_leaf=2, random_state=random_state, n_jobs=-1
)),
"gradient_boosting": lambda: _tabular_pipeline(GradientBoostingRegressor(
n_estimators=250, learning_rate=0.04, max_depth=3, loss="huber",
random_state=random_state,
)),
"random_forest": lambda: _tabular_pipeline(RandomForestRegressor(
n_estimators=300, min_samples_leaf=2, max_features=0.8,
random_state=random_state, n_jobs=-1,
)),
"xgboost": lambda: _tabular_pipeline(XGBRegressor(
n_estimators=400, learning_rate=0.04, max_depth=5, subsample=0.85,
colsample_bytree=0.85, reg_lambda=1.0, objective="reg:squarederror",
tree_method="hist", random_state=random_state, n_jobs=-1,
)),
"lightgbm": lambda: _tabular_pipeline(LGBMRegressor(
n_estimators=400, learning_rate=0.04, num_leaves=31, subsample=0.85,
colsample_bytree=0.85, reg_lambda=1.0, random_state=random_state,
verbosity=-1, n_jobs=-1,
)),
"svr": lambda: _tabular_pipeline(SVR(C=10.0, epsilon=0.1), scale=True),
"ridge": lambda: _tabular_pipeline(Ridge(alpha=1.0), scale=True),
"knn": lambda: _tabular_pipeline(KNeighborsRegressor(
n_neighbors=7, weights="distance", p=2, n_jobs=-1
), scale=True),
}
def _ensemble_predictions(
val_predictions: dict[str, np.ndarray],
test_predictions: dict[str, np.ndarray],
y_val: np.ndarray,
train_predictions: dict[str, np.ndarray],
) -> tuple[dict[str, np.ndarray], dict[str, np.ndarray]]:
names = list(val_predictions)
val_matrix = np.column_stack([val_predictions[name] for name in names])
test_matrix = np.column_stack([test_predictions[name] for name in names])
train_matrix = np.column_stack([train_predictions[name] for name in names])
errors = np.mean(np.abs(val_matrix - y_val[:, None]), axis=0)
inverse = 1.0 / np.maximum(errors, 1e-8)
weights = inverse / inverse.sum()
meta = TransformedTargetRegressor(regressor=Ridge(alpha=1.0))
meta.fit(val_matrix, y_val)
test_ensembles = {
"stacking_ensemble": meta.predict(test_matrix),
"weighted_ensemble": test_matrix @ weights,
}
# These are in-sample base-model diagnostics, not a second model-selection
# criterion. The ensemble weights and meta-model still use validation data.
train_ensembles = {
"stacking_ensemble": meta.predict(train_matrix),
"weighted_ensemble": train_matrix @ weights,
}
return test_ensembles, train_ensembles
def run_grouped_tabular_benchmark(
frame: pd.DataFrame,
*,
dataset_name: str,
n_splits: int = 5,
seeds: tuple[int, ...] = (17, 42, 2026),
feature_cols: list[str] = FEATURE_COLS_V3,
) -> tuple[pd.DataFrame, pd.DataFrame]:
"""Evaluate tabular models and return metric and prediction tables."""
missing = set(feature_cols + ["SoH", "battery_id"]).difference(frame.columns)
if missing:
raise KeyError(f"Benchmark frame is missing: {sorted(missing)}")
X = frame[feature_cols].to_numpy(dtype=float)
y = frame["SoH"].to_numpy(dtype=float)
groups = frame["battery_id"].astype(str).to_numpy()
prediction_rows: list[dict[str, object]] = []
metric_rows: list[dict[str, object]] = []
for seed in seeds:
folds = grouped_train_val_test_folds(
groups, n_splits=n_splits, validation_fraction=0.2, random_state=seed
)
for fold, (train_idx, val_idx, test_idx) in enumerate(folds, start=1):
factories = classical_factories(seed)
val_predictions: dict[str, np.ndarray] = {}
test_predictions: dict[str, np.ndarray] = {}
train_predictions: dict[str, np.ndarray] = {}
for model_id, factory in factories.items():
model = factory()
model.fit(X[train_idx], y[train_idx])
val_predictions[model_id] = model.predict(X[val_idx])
test_pred = model.predict(X[test_idx])
test_predictions[model_id] = test_pred
train_pred = model.predict(X[train_idx])
train_predictions[model_id] = train_pred
metrics = regression_metrics(y[test_idx], test_pred, n_predictors=len(feature_cols))
train_metrics = regression_metrics(y[train_idx], train_pred, n_predictors=len(feature_cols))
metric_rows.append({
"dataset": dataset_name, "seed": seed, "fold": fold,
"model": model_id, **metrics,
"train_mae": train_metrics["mae"],
"generalization_gap_mae": metrics["mae"] - train_metrics["mae"],
})
ensemble_test, ensemble_train = _ensemble_predictions(
val_predictions, test_predictions, y[val_idx], train_predictions
)
test_predictions.update(ensemble_test)
for model_id in ("stacking_ensemble", "weighted_ensemble"):
pred = test_predictions[model_id]
metrics = regression_metrics(y[test_idx], pred, n_predictors=len(feature_cols))
train_metrics = regression_metrics(
y[train_idx], ensemble_train[model_id], n_predictors=len(feature_cols)
)
metric_rows.append({
"dataset": dataset_name, "seed": seed, "fold": fold,
"model": model_id, **metrics,
"train_mae": train_metrics["mae"],
"generalization_gap_mae": metrics["mae"] - train_metrics["mae"],
})
for model_id, pred in test_predictions.items():
for local, row_idx in enumerate(test_idx):
prediction_rows.append({
"dataset": dataset_name, "seed": seed, "fold": fold,
"model": model_id, "row_index": int(row_idx),
"battery_id": groups[row_idx],
"cycle_number": int(frame.iloc[row_idx]["cycle_number"]),
"y_true": y[row_idx], "y_pred": float(pred[local]),
"residual": float(y[row_idx] - pred[local]),
})
return pd.DataFrame(metric_rows), pd.DataFrame(prediction_rows)
def run_zero_shot_tabular(
source: pd.DataFrame,
target: pd.DataFrame,
*,
source_name: str = "NASA",
target_name: str,
random_state: int = 42,
feature_cols: list[str] = FEATURE_COLS_V3,
) -> tuple[pd.DataFrame, pd.DataFrame]:
"""Train on the complete source domain and score an untouched target domain.
Ensemble weights and the stacking meta-model are learned from source-domain
out-of-fold predictions; target labels never influence fitting.
"""
X_source = source[feature_cols].to_numpy(dtype=float)
y_source = source["SoH"].to_numpy(dtype=float)
groups = source["battery_id"].astype(str).to_numpy()
X_target = target[feature_cols].to_numpy(dtype=float)
y_target = target["SoH"].to_numpy(dtype=float)
factories = classical_factories(random_state)
names = list(factories)
oof = np.zeros((len(source), len(names)), dtype=float)
cv = GroupKFold(n_splits=min(5, np.unique(groups).size), shuffle=True, random_state=random_state)
for train_idx, val_idx in cv.split(X_source, y_source, groups):
for column, name in enumerate(names):
model = factories[name]()
model.fit(X_source[train_idx], y_source[train_idx])
oof[val_idx, column] = model.predict(X_source[val_idx])
errors = np.mean(np.abs(oof - y_source[:, None]), axis=0)
weights = (1.0 / np.maximum(errors, 1e-8))
weights /= weights.sum()
meta = Ridge(alpha=1.0).fit(oof, y_source)
target_matrix = np.zeros((len(target), len(names)), dtype=float)
predictions: dict[str, np.ndarray] = {}
for column, name in enumerate(names):
model = factories[name]()
model.fit(X_source, y_source)
pred = model.predict(X_target)
predictions[name] = pred
target_matrix[:, column] = pred
predictions["stacking_ensemble"] = meta.predict(target_matrix)
predictions["weighted_ensemble"] = target_matrix @ weights
metric_rows, prediction_rows = [], []
for name, pred in predictions.items():
metric_rows.append({
"source_dataset": source_name,
"target_dataset": target_name,
"model": name,
**regression_metrics(y_target, pred, n_predictors=len(feature_cols)),
})
for index, value in enumerate(pred):
prediction_rows.append({
"source_dataset": source_name,
"target_dataset": target_name,
"model": name,
"battery_id": str(target.iloc[index]["battery_id"]),
"cycle_number": int(target.iloc[index]["cycle_number"]),
"y_true": y_target[index],
"y_pred": float(value),
"residual": float(y_target[index] - value),
})
return pd.DataFrame(metric_rows), pd.DataFrame(prediction_rows)
|