File size: 4,804 Bytes
38bc0dc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 | import os
import sys
import pickle
import numpy as np
try:
sys.stdout.reconfigure(encoding='utf-8')
sys.stderr.reconfigure(encoding='utf-8')
except Exception:
pass
from sklearn.ensemble import RandomForestClassifier
from xgboost import XGBClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import (
accuracy_score,
precision_score,
recall_score,
f1_score,
confusion_matrix,
)
from feature_pipeline import prepare_training_data, FEATURE_SCHEMA
CSV_PATH = os.path.join(os.path.dirname(__file__), "data", "dataset.csv")
RF_PATH = os.path.join(os.path.dirname(__file__), "models", "rf_model.pkl")
XGB_PATH = os.path.join(os.path.dirname(__file__), "models", "xgb_model.pkl")
RANDOM_STATE = 42
def _evaluate(model_name, model, X_test, y_test):
y_pred = model.predict(X_test)
acc = accuracy_score(y_test, y_pred)
prec = precision_score(y_test, y_pred)
rec = recall_score(y_test, y_pred)
f1 = f1_score(y_test, y_pred)
cm = confusion_matrix(y_test, y_pred)
print(f"\n{'='*55}")
print(f" {model_name} β Evaluation Results")
print(f"{'='*55}")
print(f" Accuracy : {acc:.4f} ({acc*100:.1f}%)")
print(f" Precision : {prec:.4f}")
print(f" Recall : {rec:.4f}")
print(f" F1-Score : {f1:.4f}")
print(f"{'-'*55}")
print(f" Confusion Matrix:")
print(f" Predicted Clean Predicted Risky")
print(f" Actual Clean {cm[0][0]:<18} {cm[0][1]}")
print(f" Actual Risky {cm[1][0]:<18} {cm[1][1]}")
print(f"{'='*55}")
return acc
def _save_model(model, path):
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, 'wb') as f:
pickle.dump(model, f)
print(f" Saved β {path}")
def train_model(csv_path=CSV_PATH, rf_path=RF_PATH, xgb_path=XGB_PATH):
print("Loading and scaling dataset...")
X_scaled, y = prepare_training_data(csv_path)
print(f" Total samples : {X_scaled.shape[0]}")
print(f" Features : {X_scaled.shape[1]}")
X_train, X_test, y_train, y_test = train_test_split(
X_scaled, y,
test_size=0.2,
random_state=RANDOM_STATE,
stratify=y,
)
print(f"\n Train set : {len(X_train)} samples")
print(f" Test set : {len(X_test)} samples")
print("\nTraining Random Forest...")
rf_model = RandomForestClassifier(
n_estimators=200,
max_depth=10,
min_samples_split=4,
min_samples_leaf=2,
class_weight="balanced",
random_state=RANDOM_STATE,
)
rf_model.fit(X_train, y_train)
print(" Done!")
print("\nTraining XGBoost...")
xgb_model = XGBClassifier(
n_estimators=200,
max_depth=6,
learning_rate=0.1,
subsample=0.8,
colsample_bytree=0.8,
use_label_encoder=False,
eval_metric="logloss",
random_state=RANDOM_STATE,
verbosity=0,
)
xgb_model.fit(X_train, y_train)
print(" Done!")
print("\n\nEvaluating both models on the test set...")
rf_acc = _evaluate("Random Forest", rf_model, X_test, y_test)
xgb_acc = _evaluate("XGBoost", xgb_model, X_test, y_test)
print("\n Feature Importances β Random Forest:")
importances = rf_model.feature_importances_
ranked = sorted(zip(FEATURE_SCHEMA, importances),
key=lambda x: x[1], reverse=True)
for rank, (feature, score) in enumerate(ranked, start=1):
bar = "β" * int(score * 50)
print(f" {rank:>2}. {feature:<28} {score:.4f} {bar}")
print("\nSaving models...")
_save_model(rf_model, rf_path)
_save_model(xgb_model, xgb_path)
print("\n" + "=" * 55)
print(" MODEL COMPARISON SUMMARY")
print("=" * 55)
print(f" Random Forest accuracy : {rf_acc*100:.1f}%")
print(f" XGBoost accuracy : {xgb_acc*100:.1f}%")
if rf_acc >= xgb_acc:
print(f" Winner : Random Forest π")
else:
print(f" Winner : XGBoost π")
print(f"\n Both models saved to /models/ folder.")
print(f" Random Forest is used as the primary model in the Scoring API.")
print("=" * 55)
return {"rf": rf_model, "xgb": xgb_model}
def load_model(model_path=RF_PATH):
if not os.path.exists(model_path):
raise FileNotFoundError(
f"Model not found at: {model_path}\n"
f"Run train_model() first to create it."
)
with open(model_path, 'rb') as f:
model = pickle.load(f)
return model
if __name__ == "__main__":
train_model()
|