Spaces:
No application file
No application file
File size: 2,674 Bytes
82bf8c3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 | # ===============================
# Student Performance Prediction
# ===============================
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
# -------------------------------
# 1. Load Dataset (USE YOUR PATH)
# -------------------------------
data = pd.read_csv(
r"C:\Users\tmeka\Downloads\StudentPerformanceFactors.csv"
)
print("Dataset loaded:", data.shape)
# -------------------------------
# 2. Encode Categorical Columns
# -------------------------------
for col in data.columns:
if data[col].dtype == "object":
data[col] = LabelEncoder().fit_transform(data[col])
# -------------------------------
# 3. Split Features & Target
# -------------------------------
X = data.drop("Exam_Score", axis=1) # 19 features
y = data["Exam_Score"]
print("Number of input features:", X.shape[1])
print("Feature names:\n", X.columns)
# -------------------------------
# 4. Train-Test Split
# -------------------------------
X_train, X_test, y_train, y_test = train_test_split(
X, y, test_size=0.2, random_state=42
)
# -------------------------------
# 5. Train Random Forest Model
# -------------------------------
model = RandomForestRegressor(
n_estimators=100,
random_state=42
)
model.fit(X_train, y_train)
print("Model trained successfully")
# -------------------------------
# 6. Evaluate Model
# -------------------------------
y_pred = model.predict(X_test)
print("\nModel Performance:")
print("MAE:", mean_absolute_error(y_test, y_pred))
print("MSE:", mean_squared_error(y_test, y_pred))
print("R2 :", r2_score(y_test, y_pred))
# -------------------------------
# 7. Predict for New Student
# -------------------------------
# MUST PROVIDE ALL 19 FEATURES
new_student = {
"Hours_Studied": 5,
"Attendance": 90,
"Parental_Involvement": 2,
"Access_to_Resources": 2,
"Extracurricular_Activities": 1,
"Sleep_Hours": 7,
"Previous_Scores": 85,
"Motivation_Level": 2,
"Internet_Access": 1,
"Tutoring_Sessions": 1,
"Family_Income": 2,
"Teacher_Quality": 2,
"School_Type": 1,
"Peer_Influence": 1,
"Physical_Activity": 1,
"Learning_Disabilities": 0,
"Parental_Education_Level": 2,
"Distance_from_Home": 5,
"Gender": 1
}
new_student_df = pd.DataFrame([new_student])
prediction = model.predict(new_student_df)
print("\nPredicted Exam Score:", round(prediction[0], 2)) |