{ "nbformat": 4, "nbformat_minor": 0, "metadata": { "colab": { "provenance": [] }, "kernelspec": { "name": "python3", "display_name": "Python 3" }, "language_info": { "name": "python" } }, "cells": [ { "cell_type": "code", "source": [ "import os\n", "import cv2\n", "import numpy as np\n", "import pandas as pd" ], "metadata": { "id": "RRIHYeA8vZ2f" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "execution_count": null, "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "jYHkQveBL7JQ", "outputId": "7b23f722-2369-4842-c8ed-7e274bdbe589" }, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "Drive already mounted at /content/drive; to attempt to forcibly remount, call drive.mount(\"/content/drive\", force_remount=True).\n" ] } ], "source": [ "from google.colab import drive\n", "drive.mount('/content/drive')" ] }, { "cell_type": "code", "source": [ "data_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data\"" ], "metadata": { "id": "dDXJUfqX8u1f" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "#verify the folders\n", "for folder in os.listdir(data_path):\n", " print(folder, len(os.listdir(os.path.join(data_path, folder))), \"images\")" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "O53-h2jYt67_", "outputId": "f811f07c-4c56-4239-d025-e8afe0cb05dd" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenbanana 2226 images\n", "rottenoranges 1601 images\n", "rottenapples 2342 images\n", "freshbanana 1581 images\n", "freshoranges 1466 images\n", "freshapples 1697 images\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenbanana\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenbanana')\n", "\n", "df_rottenbanana = pd.DataFrame(X_list)\n", "df_rottenbanana['target'] = y_list\n", "\n", "print(\"rottenbanana shape:\", df_rottenbanana.shape)" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "wXi7RhlVufIi", "outputId": "5bafc4d8-e709-4e9a-dca8-35ad753cbc2b" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenbanana shape: (2226, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenoranges\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenoranges')\n", "\n", "df_rottenoranges = pd.DataFrame(X_list)\n", "df_rottenoranges['target'] = y_list\n", "\n", "print(\"rottenoranges shape:\", df_rottenoranges.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "5AxSMVuTy5On", "outputId": "1ed146c6-2bb7-40ef-f391-f7c4e70ac62f" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenoranges shape: (1601, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/rottenapples\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('rottenapples')\n", "\n", "df_rottenapples = pd.DataFrame(X_list)\n", "df_rottenapples['target'] = y_list\n", "\n", "print(\"rottenapples shape:\", df_rottenapples.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "xkumK75uzA5i", "outputId": "8e2e3b2b-e1bb-404a-a3fb-b9c379031195" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "rottenapples shape: (2342, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshbanana\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshbanana')\n", "\n", "df_freshbanana = pd.DataFrame(X_list)\n", "df_freshbanana['target'] = y_list\n", "\n", "print(\"freshbanana shape:\", df_freshbanana.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "sebO1Zl5zUX6", "outputId": "3b455d4c-5d9f-41e2-906d-3b5c71015d93" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshbanana shape: (1581, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshoranges\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshoranges')\n", "\n", "df_freshoranges = pd.DataFrame(X_list)\n", "df_freshoranges['target'] = y_list\n", "\n", "print(\"freshoranges shape:\", df_freshoranges.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "wLloeGpZz-fk", "outputId": "d8e6f564-a2b8-4768-b447-ec40bb9a5af7" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshoranges shape: (1466, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "folder_path = \"/content/drive/MyDrive/Colab Notebooks/ML RESUME PROJECTS/image classiication/data/freshapples\"\n", "X_list = []\n", "y_list = []\n", "\n", "for file in os.listdir(folder_path):\n", " img_path = os.path.join(folder_path, file)\n", " img = cv2.imread(img_path)\n", " img = cv2.resize(img, (64,64))\n", " img_flat = img.flatten().astype(np.uint8)\n", " X_list.append(img_flat)\n", " y_list.append('freshapples')\n", "\n", "df_freshapples = pd.DataFrame(X_list)\n", "df_freshapples['target'] = y_list\n", "\n", "print(\"freshapples shape:\", df_freshapples.shape)\n" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "2npMz_sO0Er6", "outputId": "0d7cbb85-64a9-4c53-ff52-5b8f55a1bb24" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "freshapples shape: (1697, 12289)\n" ] } ] }, { "cell_type": "code", "source": [ "df_all = pd.concat([\n", " df_rottenbanana,\n", " df_rottenoranges,\n", " df_rottenapples,\n", " df_freshbanana,\n", " df_freshoranges,\n", " df_freshapples\n", "], ignore_index=True)\n", "\n", "print(\"Merged DataFrame shape:\", df_all.shape)\n", "print(df_all['target'].value_counts())" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "bosmiLIr0Wd7", "outputId": "92d95ed4-e752-41ab-fad6-37a83c6fbee3" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "Merged DataFrame shape: (10913, 12289)\n", "target\n", "rottenapples 2342\n", "rottenbanana 2226\n", "freshapples 1697\n", "rottenoranges 1601\n", "freshbanana 1581\n", "freshoranges 1466\n", "Name: count, dtype: int64\n" ] } ] }, { "cell_type": "code", "source": [ "# Save to CSV\n", "df_all.to_csv(\"fruit_dataset.csv\", index=False)\n", "print(\"CSV saved as:\", \"fruit_dataset.csv\")" ], "metadata": { "colab": { "base_uri": "https://localhost:8080/" }, "id": "Hqs_-9A6BkpV", "outputId": "0418d9e5-dbf6-40b3-9963-60e1c2adb0a9" }, "execution_count": null, "outputs": [ { "output_type": "stream", "name": "stdout", "text": [ "CSV saved as: fruit_dataset.csv\n" ] } ] }, { "cell_type": "markdown", "source": [ "loading dataset" ], "metadata": { "id": "dL8zxlyZq0hv" } }, { "cell_type": "code", "source": [ "import pandas as pd\n", "import numpy as np" ], "metadata": { "id": "RMW9VmBqq0Sv" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "df = pd.read_csv('/content/fruit_dataset.csv')" ], "metadata": { "id": "m4vL0JnbFsQL" }, "execution_count": null, "outputs": [] }, { "cell_type": "code", "source": [ "df.head()" ], "metadata": { "id": "cVjG80lnDy4O", "colab": { "base_uri": "https://localhost:8080/", "height": 255 }, "outputId": "a3bbf90f-9487-4dee-91b3-03f215540aab" }, "execution_count": null, "outputs": [ { "output_type": "execute_result", "data": { "text/plain": [ " 0 1 2 3 4 5 6 7 8 9 ... 12279 12280 12281 \\\n", "0 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "1 255 255 255 255 255 255 254 254 254 254 ... 254 254 254 \n", "2 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "3 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "4 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 \n", "\n", " 12282 12283 12284 12285 12286 12287 target \n", "0 0 0 0 0 0 0 rottenbanana \n", "1 244 244 244 255 255 255 rottenbanana \n", "2 0 0 0 0 0 0 rottenbanana \n", "3 0 0 0 0 0 0 rottenbanana \n", "4 0 0 0 0 0 0 rottenbanana \n", "\n", "[5 rows x 12289 columns]" ], "text/html": [ "\n", "
| \n", " | 0 | \n", "1 | \n", "2 | \n", "3 | \n", "4 | \n", "5 | \n", "6 | \n", "7 | \n", "8 | \n", "9 | \n", "... | \n", "12279 | \n", "12280 | \n", "12281 | \n", "12282 | \n", "12283 | \n", "12284 | \n", "12285 | \n", "12286 | \n", "12287 | \n", "target | \n", "
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "... | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "rottenbanana | \n", "
| 1 | \n", "255 | \n", "255 | \n", "255 | \n", "255 | \n", "255 | \n", "255 | \n", "254 | \n", "254 | \n", "254 | \n", "254 | \n", "... | \n", "254 | \n", "254 | \n", "254 | \n", "244 | \n", "244 | \n", "244 | \n", "255 | \n", "255 | \n", "255 | \n", "rottenbanana | \n", "
| 2 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "... | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "rottenbanana | \n", "
| 3 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "... | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "rottenbanana | \n", "
| 4 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "... | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "0 | \n", "rottenbanana | \n", "
5 rows × 12289 columns
\n", "