File size: 6,366 Bytes
166743f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 | import time
from pathlib import Path
import joblib
import numpy as np
import pandas as pd
from sklearn.ensemble import RandomForestClassifier
from sklearn.impute import SimpleImputer
from sklearn.metrics import accuracy_score, classification_report, f1_score
from sklearn.neural_network import MLPClassifier
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.svm import SVC
from sklearn.utils import resample
from xgboost import XGBClassifier
# ==============================================================================
# 0. DIRECTORY & DATASET SETUP
# ==============================================================================
MODEL_DIR = Path("models")
MODEL_DIR.mkdir(parents=True, exist_ok=True)
DATASET_CSV = "pose_benchmark_priority1_dataset.csv"
# ==============================================================================
# 1. LOAD AND BALANCE DATASET
# ==============================================================================
print(f"[*] Loading Priority 1 dataset: {DATASET_CSV}...")
df = pd.read_csv(DATASET_CSV)
# Separate by split
raw_train_df = df[df["split"] == "train"]
test_df = df[df["split"] == "test"] # Keep original test set intact
# --- AUTOMATIC 1:1 CLASS BALANCING FOR TRAINING ---
train_fall = raw_train_df[raw_train_df["label"] == 1]
train_normal = raw_train_df[raw_train_df["label"] == 0]
min_samples = min(len(train_fall), len(train_normal))
print(
f"[*] Raw Training Counts -> Normal (0): {len(train_normal)} | Fall (1):"
f" {len(train_fall)}"
)
print(
"[*] Balancing training set to exactly"
f" {min_samples} samples per class..."
)
# Undersample both classes to min_samples to guarantee exact 1:1 match
train_fall_balanced = resample(
train_fall, replace=False, n_samples=min_samples, random_state=42
)
train_normal_balanced = resample(
train_normal, replace=False, n_samples=min_samples, random_state=42
)
# Combine and shuffle balanced dataset
train_df = (
pd.concat([train_fall_balanced, train_normal_balanced])
.sample(frac=1, random_state=42)
.reset_index(drop=True)
)
print(
f"[*] Balanced Training Set Ready: {len(train_df)} total samples (50% Fall"
" / 50% Normal)\n"
)
# Prepare Feature & Label Arrays (Dropping metadata columns)
X_train = train_df.drop(columns=["split", "label"])
y_train = train_df["label"]
X_test = test_df.drop(columns=["split", "label"])
y_test = test_df["label"]
print(f"[*] Total Input Features per Sample: {X_train.shape[1]}")
# ==============================================================================
# 2. PREPARE DATA PIPELINES (FOR NON-XGBOOST MODELS)
# ==============================================================================
imputer_scaler = Pipeline(
[("imputer", SimpleImputer(strategy="mean")), ("scaler", StandardScaler())]
)
print("[*] Preprocessing data for standard models (imputation & scaling)...")
X_train_processed = imputer_scaler.fit_transform(X_train)
X_test_processed = imputer_scaler.transform(X_test)
# ==============================================================================
# 3. DEFINE MODELS
# ==============================================================================
models = {
"XGBoost": XGBClassifier(
n_estimators=150,
max_depth=5,
learning_rate=0.03,
scale_pos_weight=1.0, # 1:1 balanced
missing=np.nan, # Native NaN handling for missing keypoints
random_state=42,
eval_metric="logloss"
),
"Random_Forest": RandomForestClassifier(n_estimators=100, max_depth=10, random_state=42, n_jobs=-1),
"SVM_(RBF)": SVC(kernel="rbf", probability=True, random_state=42),
"MLP_(Neural_Net)": MLPClassifier(
hidden_layer_sizes=(64, 32), max_iter=500, random_state=42
),
}
# ==============================================================================
# 4. TRAIN, BENCHMARK, AND EXPORT MODELS
# ==============================================================================
results = []
print("\n[*] Starting Benchmark Training on Priority 1 Dataset...\n")
# Save preprocessing pipeline inside models/
pipeline_path = MODEL_DIR / "imputer_scaler_pipeline.pkl"
joblib.dump(imputer_scaler, pipeline_path)
print(f"[*] Exported preprocessing pipeline to: {pipeline_path}")
for name, model in models.items():
print(f"\nTraining {name}...")
# XGBoost uses raw data with NaNs; others use imputed/scaled data
X_train_curr = X_train if name == "XGBoost" else X_train_processed
X_test_curr = X_test if name == "XGBoost" else X_test_processed
# Train
start_train = time.time()
model.fit(X_train_curr, y_train)
train_time = time.time() - start_train
# Export Model File into models/
model_filename = f"{name.lower().replace('_(rbf)', '').replace('_(neural_net)', '')}_priority1_fall_model.pkl"
model_path = MODEL_DIR / model_filename
joblib.dump(model, model_path)
print(f"[*] Saved trained model to: {model_path}")
# Predict & Measure Inference Latency
start_infer = time.time()
y_pred = model.predict(X_test_curr)
infer_time = time.time() - start_infer
# Calculate Latency per sample in milliseconds
latency_ms = (infer_time / len(X_test_curr)) * 1000
# Evaluate Metrics
acc = accuracy_score(y_test, y_pred)
f1 = f1_score(y_test, y_pred, average="macro")
results.append({
"Model": name,
"Accuracy": f"{acc:.4f}",
"Macro F1": f"{f1:.4f}",
"Train Time (s)": f"{train_time:.2f}",
"Inference Latency (ms)": f"{latency_ms:.4f}",
"Saved Path": str(model_path),
})
print(f"--- {name} Classification Report ---")
print(classification_report(y_test, y_pred, target_names=["Normal (0)", "Fall (1)"]))
# ==============================================================================
# 5. DISPLAY BENCHMARK SUMMARY
# ==============================================================================
results_df = pd.DataFrame(results)
print(
"\n=========================================================================================="
)
print(
" BENCHMARK RESULTS "
)
print(
"=========================================================================================="
)
print(results_df.to_string(index=False)) |