# =============================== # Student Performance Prediction # =============================== import pandas as pd from sklearn.model_selection import train_test_split from sklearn.preprocessing import LabelEncoder from sklearn.ensemble import RandomForestRegressor from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score # ------------------------------- # 1. Load Dataset (USE YOUR PATH) # ------------------------------- data = pd.read_csv( r"C:\Users\tmeka\Downloads\StudentPerformanceFactors.csv" ) print("Dataset loaded:", data.shape) # ------------------------------- # 2. Encode Categorical Columns # ------------------------------- for col in data.columns: if data[col].dtype == "object": data[col] = LabelEncoder().fit_transform(data[col]) # ------------------------------- # 3. Split Features & Target # ------------------------------- X = data.drop("Exam_Score", axis=1) # 19 features y = data["Exam_Score"] print("Number of input features:", X.shape[1]) print("Feature names:\n", X.columns) # ------------------------------- # 4. Train-Test Split # ------------------------------- X_train, X_test, y_train, y_test = train_test_split( X, y, test_size=0.2, random_state=42 ) # ------------------------------- # 5. Train Random Forest Model # ------------------------------- model = RandomForestRegressor( n_estimators=100, random_state=42 ) model.fit(X_train, y_train) print("Model trained successfully") # ------------------------------- # 6. Evaluate Model # ------------------------------- y_pred = model.predict(X_test) print("\nModel Performance:") print("MAE:", mean_absolute_error(y_test, y_pred)) print("MSE:", mean_squared_error(y_test, y_pred)) print("R2 :", r2_score(y_test, y_pred)) # ------------------------------- # 7. Predict for New Student # ------------------------------- # MUST PROVIDE ALL 19 FEATURES new_student = { "Hours_Studied": 5, "Attendance": 90, "Parental_Involvement": 2, "Access_to_Resources": 2, "Extracurricular_Activities": 1, "Sleep_Hours": 7, "Previous_Scores": 85, "Motivation_Level": 2, "Internet_Access": 1, "Tutoring_Sessions": 1, "Family_Income": 2, "Teacher_Quality": 2, "School_Type": 1, "Peer_Influence": 1, "Physical_Activity": 1, "Learning_Disabilities": 0, "Parental_Education_Level": 2, "Distance_from_Home": 5, "Gender": 1 } new_student_df = pd.DataFrame([new_student]) prediction = model.predict(new_student_df) print("\nPredicted Exam Score:", round(prediction[0], 2))