File size: 6,366 Bytes
166743f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
import time
from pathlib import Path
import joblib
import numpy as np
import pandas as pd
from sklearn.ensemble import RandomForestClassifier
from sklearn.impute import SimpleImputer
from sklearn.metrics import accuracy_score, classification_report, f1_score
from sklearn.neural_network import MLPClassifier
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.svm import SVC
from sklearn.utils import resample
from xgboost import XGBClassifier

# ==============================================================================
# 0. DIRECTORY & DATASET SETUP
# ==============================================================================
MODEL_DIR = Path("models")
MODEL_DIR.mkdir(parents=True, exist_ok=True)

DATASET_CSV = "pose_benchmark_priority1_dataset.csv"

# ==============================================================================
# 1. LOAD AND BALANCE DATASET
# ==============================================================================
print(f"[*] Loading Priority 1 dataset: {DATASET_CSV}...")
df = pd.read_csv(DATASET_CSV)

# Separate by split
raw_train_df = df[df["split"] == "train"]
test_df = df[df["split"] == "test"]  # Keep original test set intact

# --- AUTOMATIC 1:1 CLASS BALANCING FOR TRAINING ---
train_fall = raw_train_df[raw_train_df["label"] == 1]
train_normal = raw_train_df[raw_train_df["label"] == 0]

min_samples = min(len(train_fall), len(train_normal))

print(
    f"[*] Raw Training Counts -> Normal (0): {len(train_normal)} | Fall (1):"
    f" {len(train_fall)}"
)
print(
    "[*] Balancing training set to exactly"
    f" {min_samples} samples per class..."
)

# Undersample both classes to min_samples to guarantee exact 1:1 match
train_fall_balanced = resample(
    train_fall, replace=False, n_samples=min_samples, random_state=42
)
train_normal_balanced = resample(
    train_normal, replace=False, n_samples=min_samples, random_state=42
)

# Combine and shuffle balanced dataset
train_df = (
    pd.concat([train_fall_balanced, train_normal_balanced])
    .sample(frac=1, random_state=42)
    .reset_index(drop=True)
)

print(
    f"[*] Balanced Training Set Ready: {len(train_df)} total samples (50% Fall"
    " / 50% Normal)\n"
)

# Prepare Feature & Label Arrays (Dropping metadata columns)
X_train = train_df.drop(columns=["split", "label"])
y_train = train_df["label"]
X_test = test_df.drop(columns=["split", "label"])
y_test = test_df["label"]

print(f"[*] Total Input Features per Sample: {X_train.shape[1]}")

# ==============================================================================
# 2. PREPARE DATA PIPELINES (FOR NON-XGBOOST MODELS)
# ==============================================================================
imputer_scaler = Pipeline(
    [("imputer", SimpleImputer(strategy="mean")), ("scaler", StandardScaler())]
)

print("[*] Preprocessing data for standard models (imputation & scaling)...")
X_train_processed = imputer_scaler.fit_transform(X_train)
X_test_processed = imputer_scaler.transform(X_test)

# ==============================================================================
# 3. DEFINE MODELS
# ==============================================================================
models = {
    "XGBoost": XGBClassifier(
        n_estimators=150,
        max_depth=5,
        learning_rate=0.03,
        scale_pos_weight=1.0,  # 1:1 balanced
        missing=np.nan,        # Native NaN handling for missing keypoints
        random_state=42,
        eval_metric="logloss"
    ),
    "Random_Forest": RandomForestClassifier(n_estimators=100, max_depth=10, random_state=42, n_jobs=-1),
    "SVM_(RBF)": SVC(kernel="rbf", probability=True, random_state=42),
    "MLP_(Neural_Net)": MLPClassifier(
        hidden_layer_sizes=(64, 32), max_iter=500, random_state=42
    ),
}

# ==============================================================================
# 4. TRAIN, BENCHMARK, AND EXPORT MODELS
# ==============================================================================
results = []

print("\n[*] Starting Benchmark Training on Priority 1 Dataset...\n")

# Save preprocessing pipeline inside models/
pipeline_path = MODEL_DIR / "imputer_scaler_pipeline.pkl"
joblib.dump(imputer_scaler, pipeline_path)
print(f"[*] Exported preprocessing pipeline to: {pipeline_path}")

for name, model in models.items():
    print(f"\nTraining {name}...")

    # XGBoost uses raw data with NaNs; others use imputed/scaled data
    X_train_curr = X_train if name == "XGBoost" else X_train_processed
    X_test_curr = X_test if name == "XGBoost" else X_test_processed

    # Train
    start_train = time.time()
    model.fit(X_train_curr, y_train)
    train_time = time.time() - start_train

    # Export Model File into models/
    model_filename = f"{name.lower().replace('_(rbf)', '').replace('_(neural_net)', '')}_priority1_fall_model.pkl"
    model_path = MODEL_DIR / model_filename
    joblib.dump(model, model_path)
    print(f"[*] Saved trained model to: {model_path}")

    # Predict & Measure Inference Latency
    start_infer = time.time()
    y_pred = model.predict(X_test_curr)
    infer_time = time.time() - start_infer

    # Calculate Latency per sample in milliseconds
    latency_ms = (infer_time / len(X_test_curr)) * 1000

    # Evaluate Metrics
    acc = accuracy_score(y_test, y_pred)
    f1 = f1_score(y_test, y_pred, average="macro")

    results.append({
        "Model": name,
        "Accuracy": f"{acc:.4f}",
        "Macro F1": f"{f1:.4f}",
        "Train Time (s)": f"{train_time:.2f}",
        "Inference Latency (ms)": f"{latency_ms:.4f}",
        "Saved Path": str(model_path),
    })

    print(f"--- {name} Classification Report ---")
    print(classification_report(y_test, y_pred, target_names=["Normal (0)", "Fall (1)"]))

# ==============================================================================
# 5. DISPLAY BENCHMARK SUMMARY
# ==============================================================================
results_df = pd.DataFrame(results)
print(
    "\n=========================================================================================="
)
print(
    "                                   BENCHMARK RESULTS                                      "
)
print(
    "=========================================================================================="
)
print(results_df.to_string(index=False))