File size: 2,674 Bytes
82bf8c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
# ===============================
# Student Performance Prediction
# ===============================

import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score

# -------------------------------
# 1. Load Dataset (USE YOUR PATH)
# -------------------------------
data = pd.read_csv(
    r"C:\Users\tmeka\Downloads\StudentPerformanceFactors.csv"
)

print("Dataset loaded:", data.shape)

# -------------------------------
# 2. Encode Categorical Columns
# -------------------------------
for col in data.columns:
    if data[col].dtype == "object":
        data[col] = LabelEncoder().fit_transform(data[col])

# -------------------------------
# 3. Split Features & Target
# -------------------------------
X = data.drop("Exam_Score", axis=1)   # 19 features
y = data["Exam_Score"]

print("Number of input features:", X.shape[1])
print("Feature names:\n", X.columns)

# -------------------------------
# 4. Train-Test Split
# -------------------------------
X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

# -------------------------------
# 5. Train Random Forest Model
# -------------------------------
model = RandomForestRegressor(
    n_estimators=100,
    random_state=42
)
model.fit(X_train, y_train)

print("Model trained successfully")

# -------------------------------
# 6. Evaluate Model
# -------------------------------
y_pred = model.predict(X_test)

print("\nModel Performance:")
print("MAE:", mean_absolute_error(y_test, y_pred))
print("MSE:", mean_squared_error(y_test, y_pred))
print("R2 :", r2_score(y_test, y_pred))

# -------------------------------
# 7. Predict for New Student
# -------------------------------
# MUST PROVIDE ALL 19 FEATURES
new_student = {
    "Hours_Studied": 5,
    "Attendance": 90,
    "Parental_Involvement": 2,
    "Access_to_Resources": 2,
    "Extracurricular_Activities": 1,
    "Sleep_Hours": 7,
    "Previous_Scores": 85,
    "Motivation_Level": 2,
    "Internet_Access": 1,
    "Tutoring_Sessions": 1,
    "Family_Income": 2,
    "Teacher_Quality": 2,
    "School_Type": 1,
    "Peer_Influence": 1,
    "Physical_Activity": 1,
    "Learning_Disabilities": 0,
    "Parental_Education_Level": 2,
    "Distance_from_Home": 5,
    "Gender": 1
}

new_student_df = pd.DataFrame([new_student])

prediction = model.predict(new_student_df)

print("\nPredicted Exam Score:", round(prediction[0], 2))