File size: 5,380 Bytes
906c392
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
"""
Hyperparameter search table (reviewer priority #6 supporting evidence).

The reviewer questions whether all 11 models received equally disciplined
hyperparameter optimization, or whether some were left at library defaults
while others (implicitly LightGBM) were tuned. train_classifier.py's
_build_candidate_families() already runs a per-family holdout search for
every ML model family (see _select_best_candidates), and
training_results.json records each model's selected_params and
validation_auc. This script turns that into a single audit table showing,
per model: how many candidates were tried, what the search space was, and
which configuration was selected — so the answer is verifiable rather than
asserted.

Usage:
    python -m app.training.hyperparameter_table
"""

from __future__ import annotations

import csv
import json
from pathlib import Path

MODELS_DIR = Path(__file__).resolve().parents[2] / "models"
TABLES_DIR = Path(__file__).resolve().parents[3] / "docs/academic/paper/real_tables"

# Number of hand-specified candidates per family, taken directly from
# _build_candidate_families() in train_classifier.py.
_N_CANDIDATES = {
    "Logistic Regression": 4,   # C in (0.25, 0.5, 1.0, 2.0)
    "Random Forest": 3,
    "Gradient Boosting": 3,
    "SVM (RBF)": 4,
    "MLP Neural Network": 3,
    "XGBoost": 3,
    "LightGBM": 3,
}

_SEARCH_SPACE_SUMMARY = {
    "Logistic Regression": "C in {0.25, 0.5, 1.0, 2.0}; class_weight=balanced fixed",
    "Random Forest": "n_estimators in {300,450,500}; max_depth in {12,18,None}; "
                      "max_features in {sqrt,sqrt,log2}",
    "Gradient Boosting": "n_estimators in {180,200,260}; max_depth in {2,3,4}; "
                          "learning_rate in {0.04,0.05,0.07}",
    "SVM (RBF)": "C in {1.0,3.0,6.0,10.0}; gamma in {scale,scale,0.02,0.05}",
    "MLP Neural Network": "hidden_layer_sizes in {(128,64),(192,96,32),(256,128)}; "
                           "alpha in {5e-4,1e-3,2e-3}",
    "XGBoost": "n_estimators in {240,300,500}; max_depth in {3,4,5}; "
               "learning_rate in {0.03,0.05,0.06}",
    "LightGBM": "n_estimators in {220,300,500}; max_depth in {-1,6,8}; "
                "num_leaves in {18,24,31}; learning_rate in {0.03,0.05,0.07}",
}

# DL models: architecture is fixed per model (no per-family candidate search
# like the ML side), but all four share identical training hyperparameters
# (optimizer, LR, epochs, early stopping) — recorded here for the same
# fairness audit.
_DL_SHARED_TRAINING = {
    "optimizer": "AdamW",
    "learning_rate": "1e-3",
    "weight_decay": "1e-4",
    "lr_scheduler": "ReduceLROnPlateau(mode=max, factor=0.5, patience=5)",
    "loss": "BCEWithLogitsLoss(pos_weight=n_neg/n_pos)",
    "epochs_max": "100",
    "early_stopping_patience": "10",
    "batch_size": "64",
    "seed": "42 (+fold index)",
}


def run() -> None:
    with open(MODELS_DIR / "training_results.json", "r", encoding="utf-8") as f:
        results = json.load(f)

    rows = []
    for name in _N_CANDIDATES:
        entry = results.get(name, {})
        rows.append({
            "model": name,
            "n_candidates_evaluated": _N_CANDIDATES[name],
            "selection_method": "stratified holdout (80/20), best validation AUC wins",
            "validation_auc": entry.get("validation_auc"),
            "selected_params": json.dumps(entry.get("selected_params", {}), ensure_ascii=False),
            "selection_time_sec": entry.get("selection_time_sec"),
            "search_space": _SEARCH_SPACE_SUMMARY[name],
        })

    TABLES_DIR.mkdir(parents=True, exist_ok=True)
    ml_path = TABLES_DIR / "hyperparameter_search_table.csv"
    with open(ml_path, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=list(rows[0].keys()))
        writer.writeheader()
        writer.writerows(rows)

    print(f"ML hyperparameter search table written: {ml_path}")
    for r in rows:
        print(f"  {r['model']:22s} candidates={r['n_candidates_evaluated']} "
              f"val_auc={r['validation_auc']} time={r['selection_time_sec']}s")

    dl_path = TABLES_DIR / "dl_training_hyperparameters.csv"
    with open(dl_path, "w", newline="", encoding="utf-8") as f:
        writer = csv.writer(f)
        writer.writerow(["hyperparameter", "value"])
        writer.writerow(["note", "shared across all 4 DL architectures (Deep MLP, 1D-CNN, "
                                  "Residual MLP, Attention MLP) — architecture differs, "
                                  "training recipe does not"])
        for k, v in _DL_SHARED_TRAINING.items():
            writer.writerow([k, v])

    print(f"\nDL shared training hyperparameters written: {dl_path}")
    print(
        "\nNOTE: DL models do not go through a per-family candidate search "
        "like the ML models — each of the 4 DL architectures is trained "
        "once with an identical, fixed training recipe (see dl_training_"
        "hyperparameters.csv). This is architecturally different fairness "
        "(same recipe, different architecture) vs. the ML side (same "
        "architecture family, tuned hyperparameters) — both are internally "
        "consistent, but the paper should state this distinction explicitly "
        "rather than implying all 11 models went through identical tuning."
    )


if __name__ == "__main__":
    run()