{ "nbformat": 4, "nbformat_minor": 0, "metadata": { "colab": { "provenance": [] }, "kernelspec": { "name": "python3", "display_name": "Python 3" }, "language_info": { "name": "python" } }, "cells": [ { "cell_type": "code", "source": [ "import os\n", "import cv2\n", "import numpy as np\n", "import pandas as pd" ], "metadata": { "id": "RRIHYeA8vZ2f" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "jYHkQveBL7JQ", "outputId": "7b23f722-2369-4842-c8ed-7e274bdbe589" }, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "Drive already mounted at /content/drive; to attempt to forcibly remount, call drive.mount(\"/content/drive\", force_remount=True).\n" ] } ], "source": [ "from google.colab import drive\n", "drive.mount('/content/drive')" ] }, { "cell_type": "code", "source": [ "data_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data\"" ], "metadata": { "id": "dDXJUfqX8u1f" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "#verify the folders\n", "for folder in os.listdir(data_path):\n", " print(folder, len(os.listdir(os.path.join(data_path, folder))), \"images\")" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "O53-h2jYt67_", "outputId": "f811f07c-4c56-4239-d025-e8afe0cb05dd" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenbanana 2226 images\n", "rottenoranges 1601 images\n", "rottenapples 2342 images\n", "freshbanana 1581 images\n", "freshoranges 1466 images\n", "freshapples 1697 images\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenbanana\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenbanana')\n", "\n", "df_rottenbanana = pd.DataFrame(X_list)\n", "df_rottenbanana['target'] = y_list\n", "\n", "print(\"rottenbanana shape:\", df_rottenbanana.shape)" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "wXi7RhlVufIi", "outputId": "5bafc4d8-e709-4e9a-dca8-35ad753cbc2b" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenbanana shape: (2226, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenoranges\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenoranges')\n", "\n", "df_rottenoranges = pd.DataFrame(X_list)\n", "df_rottenoranges['target'] = y_list\n", "\n", "print(\"rottenoranges shape:\", df_rottenoranges.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "5AxSMVuTy5On", "outputId": "1ed146c6-2bb7-40ef-f391-f7c4e70ac62f" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenoranges shape: (1601, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenapples\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenapples')\n", "\n", "df_rottenapples = pd.DataFrame(X_list)\n", "df_rottenapples['target'] = y_list\n", "\n", "print(\"rottenapples shape:\", df_rottenapples.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "xkumK75uzA5i", "outputId": "8e2e3b2b-e1bb-404a-a3fb-b9c379031195" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenapples shape: (2342, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshbanana\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshbanana')\n", "\n", "df_freshbanana = pd.DataFrame(X_list)\n", "df_freshbanana['target'] = y_list\n", "\n", "print(\"freshbanana shape:\", df_freshbanana.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "sebO1Zl5zUX6", "outputId": "3b455d4c-5d9f-41e2-906d-3b5c71015d93" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshbanana shape: (1581, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshoranges\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshoranges')\n", "\n", "df_freshoranges = pd.DataFrame(X_list)\n", "df_freshoranges['target'] = y_list\n", "\n", "print(\"freshoranges shape:\", df_freshoranges.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "wLloeGpZz-fk", "outputId": "d8e6f564-a2b8-4768-b447-ec40bb9a5af7" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshoranges shape: (1466, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshapples\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshapples')\n", "\n", "df_freshapples = pd.DataFrame(X_list)\n", "df_freshapples['target'] = y_list\n", "\n", "print(\"freshapples shape:\", df_freshapples.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "2npMz_sO0Er6", "outputId": "0d7cbb85-64a9-4c53-ff52-5b8f55a1bb24" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshapples shape: (1697, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "df_all = pd.concat([\n", " df_rottenbanana,\n", " df_rottenoranges,\n", " df_rottenapples,\n", " df_freshbanana,\n", " df_freshoranges,\n", " df_freshapples\n", "], ignore_index=True)\n", "\n", "print(\"Merged DataFrame shape:\", df_all.shape)\n", "print(df_all['target'].value_counts())" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "bosmiLIr0Wd7", "outputId": "92d95ed4-e752-41ab-fad6-37a83c6fbee3" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "Merged DataFrame shape: (10913, 12289)\n", "target\n", "rottenapples 2342\n", "rottenbanana 2226\n", "freshapples 1697\n", "rottenoranges 1601\n", "freshbanana 1581\n", "freshoranges 1466\n", "Name: count, dtype: int64\n" ] } ] }, { "cell_type": "code", "source": [ "# Save to CSV\n", "df_all.to_csv(\"fruit_dataset.csv\", index=False)\n", "print(\"CSV saved as:\", \"fruit_dataset.csv\")" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "Hqs_-9A6BkpV", "outputId": "0418d9e5-dbf6-40b3-9963-60e1c2adb0a9" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "CSV saved as: fruit_dataset.csv\n" ] } ] }, { "cell_type": "markdown", "source": [ "loading dataset" ], "metadata": { "id": "dL8zxlyZq0hv" } }, { "cell_type": "code", "source": [ "import pandas as pd\n", "import numpy as np" ], "metadata": { "id": "RMW9VmBqq0Sv" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "df = pd.read_csv('/content/fruit_dataset.csv')" ], "metadata": { "id": "m4vL0JnbFsQL" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "df.head()" ], "metadata": { "id": "cVjG80lnDy4O", "colab": { "base_uri": "https://localhost:8080/", "height": 255 }, "outputId": "a3bbf90f-9487-4dee-91b3-03f215540aab" }, "execution_count": null, "outputs": [ { "output_type": "execute_result", "data": { "text/plain": [ " 0 1 2 3 4 5 6 7 8 9 ... 12279 12280 12281 \\\n", "0 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "1 255 255 255 255 255 255 254 254 254 254 ... 254 254 254 \n", "2 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "3 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "4 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "\n", " 12282 12283 12284 12285 12286 12287 target \n", "0 0 0 0 0 0 0 rottenbanana \n", "1 244 244 244 255 255 255 rottenbanana \n", "2 0 0 0 0 0 0 rottenbanana \n", "3 0 0 0 0 0 0 rottenbanana \n", "4 0 0 0 0 0 0 rottenbanana \n", "\n", "[5 rows x 12289 columns]" ], "text/html": [ "\n", "
\n", "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
0123456789...122791228012281122821228312284122851228612287target
00000000000...000000000rottenbanana
1255255255255255255254254254254...254254254244244244255255255rottenbanana
20000000000...000000000rottenbanana
30000000000...000000000rottenbanana
40000000000...000000000rottenbanana
\n", "

5 rows × 12289 columns

\n", "
\n", "
\n", "\n", "
\n", " \n", "\n", " \n", "\n", " \n", "
\n", "\n", "\n", "
\n", " \n", "\n", "\n", "\n", " \n", "
\n", "\n", "
\n", "
\n" ], "application/vnd.google.colaboratory.intrinsic+json": { "type": "dataframe", "variable_name": "df" } }, "metadata": {}, "execution_count": 16 } ] }, { "cell_type": "code", "source": [ "print(df['target'].value_counts())" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "0fgwuVGcsIKa", "outputId": "ddc13cf6-c30a-443e-d105-54d2c2e3886a" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "target\n", "rottenapples 2342\n", "rottenbanana 2226\n", "freshapples 1697\n", "rottenoranges 1601\n", "freshbanana 1581\n", "freshoranges 1466\n", "Name: count, dtype: int64\n" ] } ] }, { "cell_type": "code", "source": [ "# Features\n", "X = df.drop(\"target\", axis=1)\n", "# Target\n", "y = df[\"target\"]\n", "print(X.shape)\n", "print(y.shape)" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "jmM8wZxtrFxz", "outputId": "47dd5d41-7f3e-4fef-c164-66cb82abaefc" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "(10913, 12288)\n", "(10913,)\n" ] } ] }, { "cell_type": "markdown", "source": [ "# **Encode target**" ], "metadata": { "id": "ww0O5SqiraKP" } }, { "cell_type": "code", "source": [ "from sklearn.preprocessing import LabelEncoder\n", "\n", "# Create encoder\n", "le = LabelEncoder()\n", "\n", "# Fit on y and transform\n", "y_enc = le.fit_transform(y)\n", "\n", "# Check mapping\n", "label_mapping = dict(zip(le.classes_, le.transform(le.classes_)))\n", "print(\"Label mapping:\", label_mapping)\n", "\n", "print(\"Encoded y shape:\", y_enc.shape)\n", "print(\"First 10 encoded labels:\", y_enc[:10])" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "b1uiyJxWrUP5", "outputId": "4c6e6529-e877-4057-a2a8-998c27db8275" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "Label mapping: {'freshapples': np.int64(0), 'freshbanana': np.int64(1), 'freshoranges': np.int64(2), 'rottenapples': np.int64(3), 'rottenbanana': np.int64(4), 'rottenoranges': np.int64(5)}\n", "Encoded y shape: (10913,)\n", "First 10 encoded labels: [4 4 4 4 4 4 4 4 4 4]\n" ] } ] }, { "cell_type": "markdown", "source": [ "# **Train-test split**" ], "metadata": { "id": "jVlICnkKrfGT" } }, { "cell_type": "code", "source": [ "from sklearn.model_selection import train_test_split\n", "\n", "X_train, X_test, y_train, y_test = train_test_split(X, y_enc, test_size=0.2, random_state=42, stratify=y_enc)\n", "print(X_train.shape)\n", "print(X_test.shape)\n", "print(y_train.shape)\n", "print(y_test.shape)" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "dN0vUxG8rdn4", "outputId": "c3f04af5-493f-4724-87e3-3896c03351af" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "(8730, 12288)\n", "(2183, 12288)\n", "(8730,)\n", "(2183,)\n" ] } ] }, { "cell_type": "markdown", "source": [ "Scale the data and reduce dimensionality" ], "metadata": { "id": "5mG4wNmF3mJv" } }, { "cell_type": "code", "source": [ "from sklearn.preprocessing import LabelEncoder, StandardScaler\n", "# Fit scaler on training data\n", "scaler = StandardScaler()\n", "X_train_scaled = scaler.fit_transform(X_train)\n", "X_test_scaled = scaler.transform(X_test)" ], "metadata": { "id": "JkMhNWcA3riD" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "from sklearn.decomposition import PCA\n", "# Fit PCA on training data\n", "pca = PCA(n_components=0.95)\n", "X_train = pca.fit_transform(X_train_scaled)\n", "X_test = pca.transform(X_test_scaled)" ], "metadata": { "id": "uqETOPY_4Al0" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "# Save train set\n", "df_train = pd.DataFrame(X_train)\n", "df_train[\"target\"] = y_train\n", "df_train.to_csv(\"X_train_pca.csv\", index=False)\n", "\n", "# Save test set\n", "df_test = pd.DataFrame(X_test)\n", "df_test[\"target\"] = y_test\n", "df_test.to_csv(\"X_test_pca.csv\", index=False)\n", "\n", "print(\"PCA-transformed train & test saved to CSV\")" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "qH5JIDWy9SpQ", "outputId": "a37870ff-ce42-40e4-fc0c-8d749f49798e" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "PCA-transformed train & test saved to CSV\n" ] } ] }, { "cell_type": "code", "source": [ "# ML utilities\n", "from sklearn.pipeline import Pipeline\n", "from sklearn.metrics import accuracy_score, confusion_matrix, classification_report\n", "\n", "# Classifiers\n", "from sklearn.neighbors import KNeighborsClassifier\n", "from sklearn.naive_bayes import GaussianNB\n", "from sklearn.tree import DecisionTreeClassifier\n", "from sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier\n", "from sklearn.linear_model import LogisticRegression\n", "from sklearn.svm import SVC\n", "import xgboost as xgb" ], "metadata": { "id": "OnCnSxe_rsIf" }, "execution_count": null, "outputs": [] }, { "cell_type": "markdown", "source": [ "KNN" ], "metadata": { "id": "NgBQ3EYj2ZNw" } }, { "cell_type": "code", "source": [ "knn = KNeighborsClassifier(n_neighbors=5)\n", "knn.fit(X_train, y_train)\n", "y_pred = knn.predict(X_test)\n", "\n", "print(\"=== KNN ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "X4exR8G32Xb8", "outputId": "153022d2-bc2e-47d6-f7fb-143ddb25e8dc" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== KNN ===\n", "Accuracy: 0.7998167659184608\n", "Confusion Matrix:\n", " [[288 0 10 42 0 0]\n", " [ 5 288 2 12 3 6]\n", " [ 5 9 262 14 0 3]\n", " [ 36 1 51 374 0 7]\n", " [ 5 37 2 51 340 10]\n", " [ 10 13 34 69 0 194]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.83 0.85 0.84 340\n", " 1 0.83 0.91 0.87 316\n", " 2 0.73 0.89 0.80 293\n", " 3 0.67 0.80 0.73 469\n", " 4 0.99 0.76 0.86 445\n", " 5 0.88 0.61 0.72 320\n", "\n", " accuracy 0.80 2183\n", " macro avg 0.82 0.80 0.80 2183\n", "weighted avg 0.82 0.80 0.80 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "Naive Bayes" ], "metadata": { "id": "eFmZGDQV8cna" } }, { "cell_type": "code", "source": [ "nb = GaussianNB()\n", "nb.fit(X_train, y_train)\n", "y_pred = nb.predict(X_test)\n", "\n", "print(\"=== Naive Bayes ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "upieyjX62ck2", "outputId": "87170015-b901-42df-eda0-f468d8105021" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== Naive Bayes ===\n", "Accuracy: 0.5451213925790197\n", "Confusion Matrix:\n", " [[121 11 3 144 51 10]\n", " [ 5 143 7 27 49 85]\n", " [ 8 33 111 105 27 9]\n", " [ 62 21 16 336 14 20]\n", " [ 32 82 0 16 298 17]\n", " [ 16 24 16 60 23 181]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.50 0.36 0.41 340\n", " 1 0.46 0.45 0.45 316\n", " 2 0.73 0.38 0.50 293\n", " 3 0.49 0.72 0.58 469\n", " 4 0.65 0.67 0.66 445\n", " 5 0.56 0.57 0.56 320\n", "\n", " accuracy 0.55 2183\n", " macro avg 0.56 0.52 0.53 2183\n", "weighted avg 0.56 0.55 0.54 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "Decision Tree" ], "metadata": { "id": "D9Pzm77r8kin" } }, { "cell_type": "code", "source": [ "dt = DecisionTreeClassifier(random_state=42)\n", "dt.fit(X_train, y_train)\n", "y_pred = dt.predict(X_test)\n", "\n", "print(\"=== Decision Tree ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "AnVfOKH_8f9B", "outputId": "f3cfd65c-695d-4c23-b685-a704fc29e2f0" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== Decision Tree ===\n", "Accuracy: 0.6775080164910673\n", "Confusion Matrix:\n", " [[239 7 9 54 10 21]\n", " [ 6 255 16 11 14 14]\n", " [ 25 10 181 34 3 40]\n", " [ 58 10 43 279 21 58]\n", " [ 12 13 6 28 343 43]\n", " [ 22 13 20 60 23 182]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.66 0.70 0.68 340\n", " 1 0.83 0.81 0.82 316\n", " 2 0.66 0.62 0.64 293\n", " 3 0.60 0.59 0.60 469\n", " 4 0.83 0.77 0.80 445\n", " 5 0.51 0.57 0.54 320\n", "\n", " accuracy 0.68 2183\n", " macro avg 0.68 0.68 0.68 2183\n", "weighted avg 0.68 0.68 0.68 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "Random Forest" ], "metadata": { "id": "3TGFtiKF8ywK" } }, { "cell_type": "code", "source": [ "rf = RandomForestClassifier(random_state=42)\n", "rf.fit(X_train, y_train)\n", "y_pred = rf.predict(X_test)\n", "\n", "print(\"=== Random Forest ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "EazoOQyS8ySB", "outputId": "f72620ea-4ed4-424a-a857-d32332b61c85" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== Random Forest ===\n", "Accuracy: 0.8176820888685296\n", "Confusion Matrix:\n", " [[262 1 6 59 6 6]\n", " [ 5 271 2 11 20 7]\n", " [ 4 3 222 43 16 5]\n", " [ 22 1 16 407 15 8]\n", " [ 1 5 0 6 425 8]\n", " [ 10 14 11 65 22 198]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.86 0.77 0.81 340\n", " 1 0.92 0.86 0.89 316\n", " 2 0.86 0.76 0.81 293\n", " 3 0.69 0.87 0.77 469\n", " 4 0.84 0.96 0.90 445\n", " 5 0.85 0.62 0.72 320\n", "\n", " accuracy 0.82 2183\n", " macro avg 0.84 0.80 0.81 2183\n", "weighted avg 0.83 0.82 0.82 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "AdaBoost" ], "metadata": { "id": "UYzYo4Kc9GaZ" } }, { "cell_type": "code", "source": [ "ada = AdaBoostClassifier(random_state=42)\n", "ada.fit(X_train, y_train)\n", "y_pred = ada.predict(X_test)\n", "\n", "print(\"=== AdaBoost ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "6ECM3jrs8nYX", "outputId": "dd4cb104-6bd1-4d71-beea-c1b22d1c7ee4" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== AdaBoost ===\n", "Accuracy: 0.5794777828676134\n", "Confusion Matrix:\n", " [[ 88 11 15 194 13 19]\n", " [ 8 233 20 7 9 39]\n", " [ 9 4 165 93 0 22]\n", " [ 43 6 27 339 19 35]\n", " [ 22 24 2 39 327 31]\n", " [ 25 18 39 99 26 113]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.45 0.26 0.33 340\n", " 1 0.79 0.74 0.76 316\n", " 2 0.62 0.56 0.59 293\n", " 3 0.44 0.72 0.55 469\n", " 4 0.83 0.73 0.78 445\n", " 5 0.44 0.35 0.39 320\n", "\n", " accuracy 0.58 2183\n", " macro avg 0.59 0.56 0.57 2183\n", "weighted avg 0.59 0.58 0.57 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "took a little more time" ], "metadata": { "id": "9RjtfuMG9Ybb" } }, { "cell_type": "markdown", "source": [ "Gradient Boosting" ], "metadata": { "id": "YVQ8uldm9dsA" } }, { "cell_type": "code", "source": [ "gb = GradientBoostingClassifier(random_state=42)\n", "gb.fit(X_train, y_train)\n", "y_pred = gb.predict(X_test)\n", "\n", "print(\"=== Gradient Boosting ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 391 }, "id": "0JEKc1pE9LtT", "outputId": "b1cd2022-c1fd-4dd1-8bc6-cffddee55110" }, "execution_count": null, "outputs": [ { "output_type": "error", "ename": "KeyboardInterrupt", "evalue": "", "traceback": [ "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", "\u001b[0;31mKeyboardInterrupt\u001b[0m Traceback (most recent call last)", "\u001b[0;32m/tmp/ipython-input-707962318.py\u001b[0m in \u001b[0;36m\u001b[0;34m()\u001b[0m\n\u001b[1;32m 1\u001b[0m \u001b[0mgb\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mGradientBoostingClassifier\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mrandom_state\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m42\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m----> 2\u001b[0;31m \u001b[0mgb\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mfit\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mX_train\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0my_train\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 3\u001b[0m \u001b[0my_pred\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mgb\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mpredict\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mX_test\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 4\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 5\u001b[0m \u001b[0mprint\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"=== Gradient Boosting ===\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/base.py\u001b[0m in \u001b[0;36mwrapper\u001b[0;34m(estimator, *args, **kwargs)\u001b[0m\n\u001b[1;32m 1387\u001b[0m )\n\u001b[1;32m 1388\u001b[0m ):\n\u001b[0;32m-> 1389\u001b[0;31m \u001b[0;32mreturn\u001b[0m \u001b[0mfit_method\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mestimator\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 1390\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1391\u001b[0m \u001b[0;32mreturn\u001b[0m \u001b[0mwrapper\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/ensemble/_gb.py\u001b[0m in \u001b[0;36mfit\u001b[0;34m(self, X, y, sample_weight, monitor)\u001b[0m\n\u001b[1;32m 785\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 786\u001b[0m \u001b[0;31m# fit the boosting stages\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 787\u001b[0;31m n_stages = self._fit_stages(\n\u001b[0m\u001b[1;32m 788\u001b[0m \u001b[0mX_train\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 789\u001b[0m \u001b[0my_train\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/ensemble/_gb.py\u001b[0m in \u001b[0;36m_fit_stages\u001b[0;34m(self, X, y, raw_predictions, sample_weight, random_state, X_val, y_val, sample_weight_val, begin_at_stage, monitor)\u001b[0m\n\u001b[1;32m 881\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 882\u001b[0m \u001b[0;31m# fit next stage of trees\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 883\u001b[0;31m raw_predictions = self._fit_stage(\n\u001b[0m\u001b[1;32m 884\u001b[0m \u001b[0mi\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 885\u001b[0m \u001b[0mX\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/ensemble/_gb.py\u001b[0m in \u001b[0;36m_fit_stage\u001b[0;34m(self, i, X, y, raw_predictions, sample_weight, sample_mask, random_state, X_csc, X_csr)\u001b[0m\n\u001b[1;32m 487\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 488\u001b[0m \u001b[0mX\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mX_csc\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mX_csc\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m \u001b[0;32melse\u001b[0m \u001b[0mX\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 489\u001b[0;31m tree.fit(\n\u001b[0m\u001b[1;32m 490\u001b[0m \u001b[0mX\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mneg_g_view\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0msample_weight\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0msample_weight\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcheck_input\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mFalse\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 491\u001b[0m )\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/base.py\u001b[0m in \u001b[0;36mwrapper\u001b[0;34m(estimator, *args, **kwargs)\u001b[0m\n\u001b[1;32m 1387\u001b[0m )\n\u001b[1;32m 1388\u001b[0m ):\n\u001b[0;32m-> 1389\u001b[0;31m \u001b[0;32mreturn\u001b[0m \u001b[0mfit_method\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mestimator\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 1390\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1391\u001b[0m \u001b[0;32mreturn\u001b[0m \u001b[0mwrapper\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/tree/_classes.py\u001b[0m in \u001b[0;36mfit\u001b[0;34m(self, X, y, sample_weight, check_input)\u001b[0m\n\u001b[1;32m 1402\u001b[0m \"\"\"\n\u001b[1;32m 1403\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1404\u001b[0;31m super()._fit(\n\u001b[0m\u001b[1;32m 1405\u001b[0m \u001b[0mX\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1406\u001b[0m \u001b[0my\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/sklearn/tree/_classes.py\u001b[0m in \u001b[0;36m_fit\u001b[0;34m(self, X, y, sample_weight, check_input, missing_values_in_feature_mask)\u001b[0m\n\u001b[1;32m 470\u001b[0m )\n\u001b[1;32m 471\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 472\u001b[0;31m \u001b[0mbuilder\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mbuild\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mtree_\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mX\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0my\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0msample_weight\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mmissing_values_in_feature_mask\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 473\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 474\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mn_outputs_\u001b[0m \u001b[0;34m==\u001b[0m \u001b[0;36m1\u001b[0m \u001b[0;32mand\u001b[0m \u001b[0mis_classifier\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;31mKeyboardInterrupt\u001b[0m: " ] } ] }, { "cell_type": "markdown", "source": [ "took so much more time" ], "metadata": { "id": "-jU-m4XN9uyr" } }, { "cell_type": "markdown", "source": [ "XGBoost" ], "metadata": { "id": "DcCzq53-9nZP" } }, { "cell_type": "code", "source": [ "xgb_model = xgb.XGBClassifier(n_estimators=100, eval_metric='mlogloss', use_label_encoder=False, random_state=42)\n", "xgb_model.fit(X_train, y_train)\n", "y_pred = xgb_model.predict(X_test)\n", "\n", "print(\"=== XGBoost ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/", "height": 480 }, "id": "FKIx7AtR9tE1", "outputId": "716685f3-5452-409b-fc32-5da59128621a" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stderr", "text": [ "/usr/local/lib/python3.12/dist-packages/xgboost/training.py:183: UserWarning: [09:12:11] WARNING: /workspace/src/learner.cc:738: \n", "Parameters: { \"use_label_encoder\" } are not used.\n", "\n", " bst.update(dtrain, iteration=i, fobj=obj)\n" ] }, { "output_type": "error", "ename": "KeyboardInterrupt", "evalue": "", "traceback": [ "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", "\u001b[0;31mKeyboardInterrupt\u001b[0m Traceback (most recent call last)", "\u001b[0;32m/tmp/ipython-input-417072136.py\u001b[0m in \u001b[0;36m\u001b[0;34m()\u001b[0m\n\u001b[1;32m 1\u001b[0m \u001b[0mxgb_model\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mxgb\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mXGBClassifier\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mn_estimators\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m100\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0meval_metric\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'mlogloss'\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0muse_label_encoder\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mFalse\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrandom_state\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m42\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m----> 2\u001b[0;31m \u001b[0mxgb_model\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mfit\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mX_train\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0my_train\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 3\u001b[0m \u001b[0my_pred\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mxgb_model\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mpredict\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mX_test\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 4\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 5\u001b[0m \u001b[0mprint\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"=== XGBoost ===\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/xgboost/core.py\u001b[0m in \u001b[0;36minner_f\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m 727\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mk\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0marg\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mzip\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0msig\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mparameters\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0margs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 728\u001b[0m \u001b[0mkwargs\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0marg\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 729\u001b[0;31m \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 730\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 731\u001b[0m \u001b[0;32mreturn\u001b[0m \u001b[0minner_f\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/xgboost/sklearn.py\u001b[0m in \u001b[0;36mfit\u001b[0;34m(self, X, y, sample_weight, base_margin, eval_set, verbose, xgb_model, sample_weight_eval_set, base_margin_eval_set, feature_weights)\u001b[0m\n\u001b[1;32m 1681\u001b[0m )\n\u001b[1;32m 1682\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1683\u001b[0;31m self._Booster = train(\n\u001b[0m\u001b[1;32m 1684\u001b[0m \u001b[0mparams\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1685\u001b[0m \u001b[0mtrain_dmatrix\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/xgboost/core.py\u001b[0m in \u001b[0;36minner_f\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m 727\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mk\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0marg\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mzip\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0msig\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mparameters\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0margs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 728\u001b[0m \u001b[0mkwargs\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0marg\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 729\u001b[0;31m \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 730\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 731\u001b[0m \u001b[0;32mreturn\u001b[0m \u001b[0minner_f\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/xgboost/training.py\u001b[0m in \u001b[0;36mtrain\u001b[0;34m(params, dtrain, num_boost_round, evals, obj, maximize, early_stopping_rounds, evals_result, verbose_eval, xgb_model, callbacks, custom_metric)\u001b[0m\n\u001b[1;32m 181\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mcb_container\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mbefore_iteration\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mbst\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mi\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdtrain\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mevals\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 182\u001b[0m \u001b[0;32mbreak\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 183\u001b[0;31m \u001b[0mbst\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mupdate\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdtrain\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0miteration\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mi\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfobj\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mobj\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 184\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mcb_container\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mafter_iteration\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mbst\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mi\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdtrain\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mevals\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 185\u001b[0m \u001b[0;32mbreak\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", "\u001b[0;32m/usr/local/lib/python3.12/dist-packages/xgboost/core.py\u001b[0m in \u001b[0;36mupdate\u001b[0;34m(self, dtrain, iteration, fobj)\u001b[0m\n\u001b[1;32m 2245\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mfobj\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 2246\u001b[0m _check_call(\n\u001b[0;32m-> 2247\u001b[0;31m _LIB.XGBoosterUpdateOneIter(\n\u001b[0m\u001b[1;32m 2248\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mhandle\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mctypes\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mc_int\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0miteration\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdtrain\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mhandle\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 2249\u001b[0m )\n", "\u001b[0;31mKeyboardInterrupt\u001b[0m: " ] } ] }, { "cell_type": "markdown", "source": [ "Logistic Regression" ], "metadata": { "id": "cVF6t_td9oZ0" } }, { "cell_type": "code", "source": [ "lr = LogisticRegression(max_iter=1000, random_state=42)\n", "lr.fit(X_train, y_train)\n", "y_pred = lr.predict(X_test)\n", "\n", "print(\"=== Logistic Regression ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "3Hyrqs5V9trq", "outputId": "b11970af-c5a8-4b1c-f2d8-949add167423" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== Logistic Regression ===\n", "Accuracy: 0.7480531378836464\n", "Confusion Matrix:\n", " [[227 9 6 67 2 29]\n", " [ 7 283 3 3 7 13]\n", " [ 13 2 238 22 1 17]\n", " [ 68 3 32 320 12 34]\n", " [ 13 11 1 14 374 32]\n", " [ 25 8 14 55 27 191]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.64 0.67 0.66 340\n", " 1 0.90 0.90 0.90 316\n", " 2 0.81 0.81 0.81 293\n", " 3 0.67 0.68 0.67 469\n", " 4 0.88 0.84 0.86 445\n", " 5 0.60 0.60 0.60 320\n", "\n", " accuracy 0.75 2183\n", " macro avg 0.75 0.75 0.75 2183\n", "weighted avg 0.75 0.75 0.75 2183\n", "\n" ] }, { "output_type": "stream", "name": "stderr", "text": [ "/usr/local/lib/python3.12/dist-packages/sklearn/linear_model/_logistic.py:465: ConvergenceWarning: lbfgs failed to converge (status=1):\n", "STOP: TOTAL NO. OF ITERATIONS REACHED LIMIT.\n", "\n", "Increase the number of iterations (max_iter) or scale the data as shown in:\n", " https://scikit-learn.org/stable/modules/preprocessing.html\n", "Please also refer to the documentation for alternative solver options:\n", " https://scikit-learn.org/stable/modules/linear_model.html#logistic-regression\n", " n_iter_i = _check_optimize_result(\n" ] } ] }, { "cell_type": "markdown", "source": [ "SVC" ], "metadata": { "id": "Ygj8VQJc9qix" } }, { "cell_type": "code", "source": [ "svc = SVC(random_state=42)\n", "svc.fit(X_train, y_train)\n", "y_pred = svc.predict(X_test)\n", "\n", "print(\"=== SVC ===\")\n", "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n", "print(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))\n", "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "YNUh055l9hcj", "outputId": "d134b0bb-7ed9-4880-ad85-4e82936d44ea" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "=== SVC ===\n", "Accuracy: 0.9060925332111773\n", "Confusion Matrix:\n", " [[303 0 0 31 1 5]\n", " [ 7 304 0 1 3 1]\n", " [ 4 2 269 13 0 5]\n", " [ 7 1 15 433 5 8]\n", " [ 2 1 0 6 428 8]\n", " [ 4 0 11 59 5 241]]\n", "Classification Report:\n", " precision recall f1-score support\n", "\n", " 0 0.93 0.89 0.91 340\n", " 1 0.99 0.96 0.97 316\n", " 2 0.91 0.92 0.91 293\n", " 3 0.80 0.92 0.86 469\n", " 4 0.97 0.96 0.97 445\n", " 5 0.90 0.75 0.82 320\n", "\n", " accuracy 0.91 2183\n", " macro avg 0.92 0.90 0.91 2183\n", "weighted avg 0.91 0.91 0.91 2183\n", "\n" ] } ] }, { "cell_type": "markdown", "source": [ "✅ SVC (Support Vector Classifier) is clearly the winner here (91% accuracy) — this is expected since:\n", "\n", "You scaled the data ✅\n", "\n", "You did PCA (reduces noise & keeps variance) ✅\n", "\n", "SVC works very well on high-dimensional but dense feature spaces (like PCA image features).\n", "\n", "⚡ Random Forest also did well (82%), but SVC is significantly better." ], "metadata": { "id": "i6zF3_gaAH7D" } }, { "cell_type": "code", "source": [ "import pickle\n", "\n", "# ---- Save the trained SVC model ----\n", "with open(\"svc_model.pkl\", \"wb\") as f:\n", " pickle.dump(svc, f)\n", "\n", "print(\"SVC model saved as svc_model.pkl\")\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "7uFLb8WC_697", "outputId": "5286a2d1-de14-43fa-ea67-06b590bea87d" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "SVC model saved as svc_model.pkl\n" ] } ] }, { "cell_type": "code", "source": [], "metadata": { "id": "fdPko5vsBlUv" }, "execution_count": null, "outputs": [] } ] }