File size: 4,804 Bytes
38bc0dc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
import os
import sys
import pickle
import numpy as np

try:
    sys.stdout.reconfigure(encoding='utf-8')
    sys.stderr.reconfigure(encoding='utf-8')
except Exception:
    pass

from sklearn.ensemble import RandomForestClassifier
from xgboost import XGBClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
    confusion_matrix,
)

from feature_pipeline import prepare_training_data, FEATURE_SCHEMA

CSV_PATH   = os.path.join(os.path.dirname(__file__), "data",   "dataset.csv")
RF_PATH    = os.path.join(os.path.dirname(__file__), "models", "rf_model.pkl")
XGB_PATH   = os.path.join(os.path.dirname(__file__), "models", "xgb_model.pkl")

RANDOM_STATE = 42

def _evaluate(model_name, model, X_test, y_test):
    y_pred   = model.predict(X_test)
    acc      = accuracy_score(y_test, y_pred)
    prec     = precision_score(y_test, y_pred)
    rec      = recall_score(y_test, y_pred)
    f1       = f1_score(y_test, y_pred)
    cm       = confusion_matrix(y_test, y_pred)

    print(f"\n{'='*55}")
    print(f"  {model_name} β€” Evaluation Results")
    print(f"{'='*55}")
    print(f"  Accuracy  : {acc:.4f}   ({acc*100:.1f}%)")
    print(f"  Precision : {prec:.4f}")
    print(f"  Recall    : {rec:.4f}")
    print(f"  F1-Score  : {f1:.4f}")
    print(f"{'-'*55}")
    print(f"  Confusion Matrix:")
    print(f"                    Predicted Clean    Predicted Risky")
    print(f"  Actual Clean      {cm[0][0]:<18}  {cm[0][1]}")
    print(f"  Actual Risky      {cm[1][0]:<18}  {cm[1][1]}")
    print(f"{'='*55}")

    return acc

def _save_model(model, path):
    os.makedirs(os.path.dirname(path), exist_ok=True)
    with open(path, 'wb') as f:
        pickle.dump(model, f)
    print(f"  Saved β†’ {path}")

def train_model(csv_path=CSV_PATH, rf_path=RF_PATH, xgb_path=XGB_PATH):
    print("Loading and scaling dataset...")
    X_scaled, y = prepare_training_data(csv_path)
    print(f"  Total samples : {X_scaled.shape[0]}")
    print(f"  Features      : {X_scaled.shape[1]}")

    X_train, X_test, y_train, y_test = train_test_split(
        X_scaled, y,
        test_size=0.2,
        random_state=RANDOM_STATE,
        stratify=y,
    )
    print(f"\n  Train set : {len(X_train)} samples")
    print(f"  Test set  : {len(X_test)} samples")

    print("\nTraining Random Forest...")
    rf_model = RandomForestClassifier(
        n_estimators=200,
        max_depth=10,
        min_samples_split=4,
        min_samples_leaf=2,
        class_weight="balanced",
        random_state=RANDOM_STATE,
    )
    rf_model.fit(X_train, y_train)
    print("  Done!")

    print("\nTraining XGBoost...")
    xgb_model = XGBClassifier(
        n_estimators=200,
        max_depth=6,
        learning_rate=0.1,
        subsample=0.8,
        colsample_bytree=0.8,
        use_label_encoder=False,
        eval_metric="logloss",
        random_state=RANDOM_STATE,
        verbosity=0,
    )
    xgb_model.fit(X_train, y_train)
    print("  Done!")

    print("\n\nEvaluating both models on the test set...")

    rf_acc  = _evaluate("Random Forest", rf_model,  X_test, y_test)
    xgb_acc = _evaluate("XGBoost",       xgb_model, X_test, y_test)

    print("\n  Feature Importances β€” Random Forest:")
    importances = rf_model.feature_importances_
    ranked = sorted(zip(FEATURE_SCHEMA, importances),
                    key=lambda x: x[1], reverse=True)
    for rank, (feature, score) in enumerate(ranked, start=1):
        bar = "β–ˆ" * int(score * 50)
        print(f"  {rank:>2}. {feature:<28} {score:.4f}  {bar}")

    print("\nSaving models...")
    _save_model(rf_model,  rf_path)
    _save_model(xgb_model, xgb_path)

    print("\n" + "=" * 55)
    print("  MODEL COMPARISON SUMMARY")
    print("=" * 55)
    print(f"  Random Forest accuracy : {rf_acc*100:.1f}%")
    print(f"  XGBoost accuracy       : {xgb_acc*100:.1f}%")

    if rf_acc >= xgb_acc:
        print(f"  Winner                 : Random Forest πŸ†")
    else:
        print(f"  Winner                 : XGBoost πŸ†")

    print(f"\n  Both models saved to /models/ folder.")
    print(f"  Random Forest is used as the primary model in the Scoring API.")
    print("=" * 55)

    return {"rf": rf_model, "xgb": xgb_model}

def load_model(model_path=RF_PATH):
    if not os.path.exists(model_path):
        raise FileNotFoundError(
            f"Model not found at: {model_path}\n"
            f"Run train_model() first to create it."
        )
    with open(model_path, 'rb') as f:
        model = pickle.load(f)
    return model

if __name__ == "__main__":
    train_model()