{
 "cells": [
  {
   "cell_type": "markdown",
   "id": "5a6cdf5f-13e9-423a-ad26-7616c1f3228b",
   "metadata": {},
   "source": [
    "\n",
    "# **Practical Machine Learning and Data Exploration**\n",
    "## Assignment 2 - **Breast Cancer Dataset**\n",
    "### Student: Sara Ismail\n",
    "\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "b472c786-a388-48eb-aca1-afe0a2f4e494",
   "metadata": {},
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "import numpy as np\n",
    "from sklearn.datasets import load_iris\n",
    "from sklearn.feature_selection import SelectKBest, chi2, RFE\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.model_selection import train_test_split\n",
    "from sklearn.model_selection import GridSearchCV\n",
    "from sklearn.preprocessing import StandardScaler, LabelEncoder\n",
    "from sklearn.tree import DecisionTreeClassifier\n",
    "from sklearn.svm import SVC\n",
    "from sklearn.naive_bayes import GaussianNB\n",
    "from sklearn.neighbors import KNeighborsClassifier\n",
    "from sklearn.metrics import (\n",
    " confusion_matrix,\n",
    " accuracy_score,\n",
    " precision_score,\n",
    " recall_score,\n",
    " f1_score,\n",
    " classification_report\n",
    ")\n",
    "import matplotlib.pyplot as plt\n",
    "import seaborn as sns"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "57ae328a-7fe0-4192-8a64-d8473e18d23b",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>Age</th>\n",
       "      <th>Race</th>\n",
       "      <th>Marital Status</th>\n",
       "      <th>T Stage</th>\n",
       "      <th>N Stage</th>\n",
       "      <th>6th Stage</th>\n",
       "      <th>differentiate</th>\n",
       "      <th>Grade</th>\n",
       "      <th>A Stage</th>\n",
       "      <th>Tumor Size</th>\n",
       "      <th>Estrogen Status</th>\n",
       "      <th>Progesterone Status</th>\n",
       "      <th>Regional Node Examined</th>\n",
       "      <th>Reginol Node Positive</th>\n",
       "      <th>Survival Months</th>\n",
       "      <th>Status</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>68</td>\n",
       "      <td>White</td>\n",
       "      <td>Married</td>\n",
       "      <td>T1</td>\n",
       "      <td>N1</td>\n",
       "      <td>IIA</td>\n",
       "      <td>Poorly differentiated</td>\n",
       "      <td>3</td>\n",
       "      <td>Regional</td>\n",
       "      <td>4</td>\n",
       "      <td>Positive</td>\n",
       "      <td>Positive</td>\n",
       "      <td>24</td>\n",
       "      <td>1</td>\n",
       "      <td>60</td>\n",
       "      <td>Alive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>50</td>\n",
       "      <td>White</td>\n",
       "      <td>Married</td>\n",
       "      <td>T2</td>\n",
       "      <td>N2</td>\n",
       "      <td>IIIA</td>\n",
       "      <td>Moderately differentiated</td>\n",
       "      <td>2</td>\n",
       "      <td>Regional</td>\n",
       "      <td>35</td>\n",
       "      <td>Positive</td>\n",
       "      <td>Positive</td>\n",
       "      <td>14</td>\n",
       "      <td>5</td>\n",
       "      <td>62</td>\n",
       "      <td>Alive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>58</td>\n",
       "      <td>White</td>\n",
       "      <td>Divorced</td>\n",
       "      <td>T3</td>\n",
       "      <td>N3</td>\n",
       "      <td>IIIC</td>\n",
       "      <td>Moderately differentiated</td>\n",
       "      <td>2</td>\n",
       "      <td>Regional</td>\n",
       "      <td>63</td>\n",
       "      <td>Positive</td>\n",
       "      <td>Positive</td>\n",
       "      <td>14</td>\n",
       "      <td>7</td>\n",
       "      <td>75</td>\n",
       "      <td>Alive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>3</th>\n",
       "      <td>58</td>\n",
       "      <td>White</td>\n",
       "      <td>Married</td>\n",
       "      <td>T1</td>\n",
       "      <td>N1</td>\n",
       "      <td>IIA</td>\n",
       "      <td>Poorly differentiated</td>\n",
       "      <td>3</td>\n",
       "      <td>Regional</td>\n",
       "      <td>18</td>\n",
       "      <td>Positive</td>\n",
       "      <td>Positive</td>\n",
       "      <td>2</td>\n",
       "      <td>1</td>\n",
       "      <td>84</td>\n",
       "      <td>Alive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>4</th>\n",
       "      <td>47</td>\n",
       "      <td>White</td>\n",
       "      <td>Married</td>\n",
       "      <td>T2</td>\n",
       "      <td>N1</td>\n",
       "      <td>IIB</td>\n",
       "      <td>Poorly differentiated</td>\n",
       "      <td>3</td>\n",
       "      <td>Regional</td>\n",
       "      <td>41</td>\n",
       "      <td>Positive</td>\n",
       "      <td>Positive</td>\n",
       "      <td>3</td>\n",
       "      <td>1</td>\n",
       "      <td>50</td>\n",
       "      <td>Alive</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "   Age   Race Marital Status T Stage  N Stage 6th Stage  \\\n",
       "0   68  White        Married       T1      N1       IIA   \n",
       "1   50  White        Married       T2      N2      IIIA   \n",
       "2   58  White       Divorced       T3      N3      IIIC   \n",
       "3   58  White        Married       T1      N1       IIA   \n",
       "4   47  White        Married       T2      N1       IIB   \n",
       "\n",
       "               differentiate Grade   A Stage  Tumor Size Estrogen Status  \\\n",
       "0      Poorly differentiated     3  Regional           4        Positive   \n",
       "1  Moderately differentiated     2  Regional          35        Positive   \n",
       "2  Moderately differentiated     2  Regional          63        Positive   \n",
       "3      Poorly differentiated     3  Regional          18        Positive   \n",
       "4      Poorly differentiated     3  Regional          41        Positive   \n",
       "\n",
       "  Progesterone Status  Regional Node Examined  Reginol Node Positive  \\\n",
       "0            Positive                      24                      1   \n",
       "1            Positive                      14                      5   \n",
       "2            Positive                      14                      7   \n",
       "3            Positive                       2                      1   \n",
       "4            Positive                       3                      1   \n",
       "\n",
       "   Survival Months Status  \n",
       "0               60  Alive  \n",
       "1               62  Alive  \n",
       "2               75  Alive  \n",
       "3               84  Alive  \n",
       "4               50  Alive  "
      ]
     },
     "execution_count": 2,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "# Load dataset\n",
    "df = pd.read_csv(r\"C:\\Users\\someO\\Desktop\\Master\\Practical_Machine_Learning_and_Data_Exploration/Assignment 2/Breast_Cancer.csv\")\n",
    "# Preview first 5 rows\n",
    "df.head()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "62a2b3f1-4b21-42fc-9b41-d9ca0ae1ebc3",
   "metadata": {},
   "source": [
    "### Number of Records per Class"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "bd081ee5-54f6-4fef-bd16-03d08a54b61e",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "Status\n",
       "Alive    3408\n",
       "Dead      616\n",
       "Name: count, dtype: int64"
      ]
     },
     "execution_count": 3,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df['Status'].value_counts()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "1d3fe324-04e1-4d23-ae23-59c444936ead",
   "metadata": {},
   "source": [
    "# Identify Data Types of Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "id": "8a93907a-fa84-47f2-a6d1-9b690360690e",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "Age                        int64\n",
       "Race                      object\n",
       "Marital Status            object\n",
       "T Stage                   object\n",
       "N Stage                   object\n",
       "6th Stage                 object\n",
       "differentiate             object\n",
       "Grade                     object\n",
       "A Stage                   object\n",
       "Tumor Size                 int64\n",
       "Estrogen Status           object\n",
       "Progesterone Status       object\n",
       "Regional Node Examined     int64\n",
       "Reginol Node Positive      int64\n",
       "Survival Months            int64\n",
       "Status                    object\n",
       "dtype: object"
      ]
     },
     "execution_count": 4,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df.dtypes"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "c544c065-b49e-462d-91e3-0df73becade1",
   "metadata": {},
   "source": [
    "# Handling missing values\n",
    "## Count of Missing Values"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "bd35ee52-bad8-4472-a92d-a4f0ce349927",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "Age                       0\n",
       "Race                      0\n",
       "Marital Status            0\n",
       "T Stage                   0\n",
       "N Stage                   0\n",
       "6th Stage                 0\n",
       "differentiate             0\n",
       "Grade                     0\n",
       "A Stage                   0\n",
       "Tumor Size                0\n",
       "Estrogen Status           0\n",
       "Progesterone Status       0\n",
       "Regional Node Examined    0\n",
       "Reginol Node Positive     0\n",
       "Survival Months           0\n",
       "Status                    0\n",
       "dtype: int64"
      ]
     },
     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df.isnull().sum()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "b70883df-689a-4ab8-873a-c1fdb7976c84",
   "metadata": {},
   "source": [
    "#### Dataset does not have any missing values, so we do not need to take any actions to modify or process the missing values."
   ]
  },
  {
   "cell_type": "markdown",
   "id": "cc05dc46-d2f1-4b9b-b307-292f26603a8c",
   "metadata": {},
   "source": [
    "# Handling Qutliers"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "2673a9f3-8f77-4264-8d11-a50d52e7efcf",
   "metadata": {},
   "source": [
    "## Detect Outliers\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "id": "b2f9e7bd-97b9-48e5-9b81-ebb1334496cd",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Sum of null values in each column:\n",
      "Age                         0\n",
      "Race                        0\n",
      "Marital Status              0\n",
      "T Stage                     0\n",
      "N Stage                     0\n",
      "6th Stage                   0\n",
      "differentiate               0\n",
      "Grade                       0\n",
      "A Stage                     0\n",
      "Tumor Size                222\n",
      "Estrogen Status             0\n",
      "Progesterone Status         0\n",
      "Regional Node Examined     72\n",
      "Reginol Node Positive     344\n",
      "Survival Months            18\n",
      "Status                      0\n",
      "dtype: int64\n"
     ]
    }
   ],
   "source": [
    "def replace_outliers_with_nan(column):\n",
    "    Q1 = column.quantile(0.25)\n",
    "    Q3 = column.quantile(0.75)\n",
    "    IQR = Q3 - Q1\n",
    "    lower_bound = Q1 - 1.5 * IQR\n",
    "    upper_bound = Q3 + 1.5 * IQR\n",
    "    # Replace outliers with NaN\n",
    "    return column.where((column >= lower_bound) & (column <= upper_bound), np.nan)\n",
    "\n",
    "\n",
    "# Replace outliers with NaN for numiric feature\n",
    "for col in df.select_dtypes(include='number').columns:\n",
    "    df[col] = replace_outliers_with_nan(df[col])\n",
    "\n",
    "# Recount missing values\n",
    "null_counts = df.isnull().sum()\n",
    "print(\"Sum of null values in each column:\")\n",
    "print(null_counts)\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "5df117d6-8c1b-49f3-a182-9937b90731ea",
   "metadata": {},
   "source": [
    "## Re-handling missing values"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "74e19818-5884-497d-af23-c25e8e148a5c",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Columns dropped: []\n",
      "---------------------------------\n",
      "Remaining columns: Index(['Age', 'Race', 'Marital Status', 'T Stage ', 'N Stage', '6th Stage',\n",
      "       'differentiate', 'Grade', 'A Stage', 'Tumor Size', 'Estrogen Status',\n",
      "       'Progesterone Status', 'Regional Node Examined',\n",
      "       'Reginol Node Positive', 'Survival Months', 'Status'],\n",
      "      dtype='object')\n",
      "---------------------------------\n",
      "Count of Missing Values\n"
     ]
    },
    {
     "data": {
      "text/plain": [
       "Age                       0\n",
       "Race                      0\n",
       "Marital Status            0\n",
       "T Stage                   0\n",
       "N Stage                   0\n",
       "6th Stage                 0\n",
       "differentiate             0\n",
       "Grade                     0\n",
       "A Stage                   0\n",
       "Tumor Size                0\n",
       "Estrogen Status           0\n",
       "Progesterone Status       0\n",
       "Regional Node Examined    0\n",
       "Reginol Node Positive     0\n",
       "Survival Months           0\n",
       "Status                    0\n",
       "dtype: int64"
      ]
     },
     "execution_count": 7,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "\n",
    "## Dropping Columns With >30% Missing Data\n",
    "\n",
    "missing_percentage = df.isnull().mean() * 100\n",
    "threshold = 30\n",
    "# Filter columns\n",
    "columns_to_drop = missing_percentage[missing_percentage > threshold].index\n",
    "\n",
    "# Drop columns\n",
    "df = df.drop(columns=columns_to_drop)\n",
    "\n",
    "# Results\n",
    "print(f\"Columns dropped: {list(columns_to_drop)}\")\n",
    "print(f\"---------------------------------\")\n",
    "print(f\"Remaining columns: {df.columns}\")\n",
    "\n",
    "#------------------------------------------------------------------------------#\n",
    "\n",
    "## Handling Nulls Based on Data Type\n",
    "\n",
    "# Fill missing values in columns with mode\n",
    "for column in df.select_dtypes(include=['float64']).columns:\n",
    " mean_value = df[column].mean() # Get the mode (most frequent value)\n",
    " df[column] = df[column].fillna(mean_value)\n",
    "\n",
    "# Fill missing values in categorical columns with mode\n",
    "for column in df.select_dtypes(include=['object']).columns:\n",
    " mode_value = df[column].mode()[0] # Get the mode (most frequent value)\n",
    " df[column] = df[column].fillna(mode_value)\n",
    "\n",
    "#Recount missing values\n",
    "print(f\"---------------------------------\")\n",
    "print(f\"Count of Missing Values\")\n",
    "df.isnull().sum()\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "e5691f2f-d4ad-4db0-a7ad-fd56a8b2a349",
   "metadata": {},
   "source": [
    "# Work with Feature Selection And Classification Models"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "f027c9fb-b243-4e8c-a6b9-edcfd84703bf",
   "metadata": {},
   "source": [
    "## Prepare Data :"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "3735b599-0fd4-4f1b-a2f2-392a7940a2fc",
   "metadata": {},
   "source": [
    "###  Convert categorical features to numerical data:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "id": "0b955500-1e67-4bcd-8e8d-4994f24a99a5",
   "metadata": {},
   "outputs": [],
   "source": [
    "label_encoders = {}\n",
    "for column in df.select_dtypes(include=['object']).columns:\n",
    " le = LabelEncoder()\n",
    " df[column] = le.fit_transform(df[column])\n",
    " label_encoders[column] = le"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "27991310-cc50-495a-bdee-be82db4da375",
   "metadata": {},
   "source": [
    "### Define features and target variable:\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "id": "1c0a1965-73be-4f6c-92e2-b38695601679",
   "metadata": {},
   "outputs": [],
   "source": [
    "X = df.drop('Status', axis=1)\n",
    "y = df['Status']"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "5a2dc314-560a-4f57-81ec-b2efc788f895",
   "metadata": {},
   "source": [
    "### Split data into training and testing sets:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 10,
   "id": "96d903e8-4354-4c0d-9b79-e069f844bc1d",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train, X_test, y_train, y_test = train_test_split(X, y,test_size=0.2, random_state=42)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "1f299d80-0f70-468c-b841-1d21871d2232",
   "metadata": {},
   "source": [
    "## Define Evaluation model:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 11,
   "id": "32829bca-a626-4cc5-88a7-7f19ac5800b1",
   "metadata": {},
   "outputs": [],
   "source": [
    "def evaluate_model(model_name, y_true, y_pred):\n",
    "    # Calculate metrics\n",
    "    accuracy = accuracy_score(y_true, y_pred)\n",
    "    precision = precision_score(y_true, y_pred, average='weighted')\n",
    "    recall = recall_score(y_true, y_pred, average='weighted')\n",
    "    f1 = f1_score(y_true, y_pred, average='weighted')\n",
    "    \n",
    "    # Output results\n",
    "    metrics = {\n",
    "        'Model Name': model_name,\n",
    "        'Accuracy': accuracy,\n",
    "        'Precision': precision,\n",
    "        'Recall': recall,\n",
    "        'F1 Score': f1,\n",
    "    }\n",
    "    \n",
    "    return metrics"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "2d2fc1a6-f2cb-40ed-a1eb-613b2c628f1e",
   "metadata": {},
   "source": [
    "# Hyper parameters And GridSearchCV\n",
    "### define one function to reuse it [cleaner code]"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 12,
   "id": "6d21ed30-0793-4a95-99c6-4595bf58ac39",
   "metadata": {},
   "outputs": [],
   "source": [
    "\n",
    "def run_my_grid_search(classifier, param_grid, X_train, y_train, X_test, y_test, model_name):\n",
    "\n",
    "    # Initialize GridSearchCV\n",
    "    grid_search = GridSearchCV(classifier,param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)\n",
    "    \n",
    "    # Fit the grid search to the data\n",
    "    grid_search.fit(X_train_selected, y_train)\n",
    "    \n",
    "    # Get the best estimator from grid search\n",
    "    best_rf_classifier = grid_search.best_estimator_\n",
    "    \n",
    "    # Make predictions using the best model\n",
    "    y_pred = best_rf_classifier.predict(X_test_selected)\n",
    "    \n",
    "    # Evaluate the model\n",
    "    evaluation_results = evaluate_model(model_name, y_test, y_pred)\n",
    "    \n",
    "    # Print the evaluation results\n",
    "    for key, value in evaluation_results.items():\n",
    "        if key == 'Classification Report':\n",
    "            print(value)\n",
    "        else:\n",
    "            print(f\"{key}: {value:.4f}\" if isinstance(value, float) else f\"{key}: \\n{value}\")\n",
    "    \n",
    "    \n",
    "    # Print the best parameters found by grid search\n",
    "    print(\"\\nBest hyperparameters found by GridSearchCV:\")\n",
    "    print(grid_search.best_params_)\n",
    "\n",
    "    "
   ]
  },
  {
   "cell_type": "markdown",
   "id": "0698faa5-ce08-4ad6-af56-d9c3371e84f4",
   "metadata": {},
   "source": [
    "# -----------------------------------------------"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "4c473daa-d0e8-4c6f-b6ca-af93b30c2a18",
   "metadata": {},
   "source": [
    "## 1. Filter Method (Chi-Square Test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "id": "c2394e21-7b58-4cfa-8e05-9b79e60060a0",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "                   Feature        Score\n",
      "14         Survival Months  4529.216268\n",
      "9               Tumor Size   300.134389\n",
      "5                6th Stage   237.010043\n",
      "4                  N Stage   208.662477\n",
      "13   Reginol Node Positive   110.204014\n",
      "3                 T Stage     55.331773\n",
      "0                      Age    19.951705\n",
      "11     Progesterone Status    17.175592\n",
      "7                    Grade    11.323282\n",
      "12  Regional Node Examined     9.976982\n",
      "10         Estrogen Status     6.718463\n",
      "2           Marital Status     2.099589\n",
      "8                  A Stage     0.755447\n",
      "6            differentiate     0.696800\n",
      "1                     Race     0.417273\n"
     ]
    }
   ],
   "source": [
    "select_feature = SelectKBest(score_func=chi2, k=8).fit(X_train, y_train)\n",
    "feature_scores = pd.DataFrame({\n",
    "    'Feature': X.columns,             \n",
    "    'Score': select_feature.scores_\n",
    "})\n",
    "\n",
    "print(feature_scores.sort_values(by='Score', ascending=False))\n",
    "\n",
    "X_train_selected = select_feature.transform(X_train)\n",
    "X_test_selected = select_feature.transform(X_test)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "6ff33b85-385a-4731-be49-a51ae72fc9a1",
   "metadata": {},
   "source": [
    "##### Applying Random Forest on selected features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 14,
   "id": "0be580a3-7379-41b6-b64d-3378a70337b6",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.9106\n",
      "Precision: 0.9060\n",
      "Recall: 0.9106\n",
      "F1 Score: 0.9004\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': 10, 'min_samples_leaf': 2, 'min_samples_split': 10, 'n_estimators': 100}\n"
     ]
    }
   ],
   "source": [
    "\n",
    "# Define the Random Forest Classifier\n",
    "rf_classifier = RandomForestClassifier(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'n_estimators': [100, 200, 300],\n",
    " 'max_depth': [None, 10, 20, 30],\n",
    " 'min_samples_split': [2, 5, 10],\n",
    " 'min_samples_leaf': [1, 2, 4]\n",
    "}\n",
    "\n",
    "run_my_grid_search(rf_classifier, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"RandomForestClassifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "100df994-c168-4de0-8d4f-0033bc04436e",
   "metadata": {},
   "source": [
    "##### Applying Logistic Regression on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "id": "af0ffd07-4d7a-4dc2-aa30-c2316ecff151",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "LogisticRegressionClassifier\n",
      "Accuracy: 0.9081\n",
      "Precision: 0.9037\n",
      "Recall: 0.9081\n",
      "F1 Score: 0.8965\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'C': 10, 'max_iter': 100, 'solver': 'liblinear'}\n"
     ]
    }
   ],
   "source": [
    "# Initialize the model\n",
    "LR = LogisticRegression(max_iter=5000, random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'C': [0.01, 0.1, 1, 10], # Regularization strength\n",
    " 'solver': ['liblinear', 'lbfgs'], # Solver type\n",
    " 'max_iter': [100, 200, 300] # Maximum iterations for convergence\n",
    "}\n",
    "\n",
    "\n",
    "run_my_grid_search(LR, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"LogisticRegressionClassifier\")\n",
    "\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "88dc1b6a-a11c-4c7d-ba06-7dfd0a124e60",
   "metadata": {},
   "source": [
    "##### Applying Decision Tree on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 16,
   "id": "bbabe801-8df0-4520-8b3d-a1cd7f832e94",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "DecisionTreeClassifier\n",
      "Accuracy: 0.9081\n",
      "Precision: 0.9023\n",
      "Recall: 0.9081\n",
      "F1 Score: 0.8981\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'criterion': 'gini', 'max_depth': 10, 'max_features': 'log2', 'min_samples_leaf': 4, 'min_samples_split': 10}\n"
     ]
    }
   ],
   "source": [
    "# Initialize the model\n",
    "DT = DecisionTreeClassifier(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'criterion': ['gini', 'entropy'], # The function to measure the qual\n",
    " 'max_depth': [None, 10, 20, 30], # The maximum depth of the tree\n",
    " 'min_samples_split': [2, 5, 10], # The minimum number of samples r\n",
    " 'min_samples_leaf': [1, 2, 4], # The minimum number of samples r\n",
    " 'max_features': [None, 'sqrt', 'log2'] # The number of features to consid\n",
    "}\n",
    "\n",
    "run_my_grid_search(DT, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"DecisionTreeClassifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "7424f8b4-40ab-4565-9c74-fd91c257cfec",
   "metadata": {},
   "source": [
    "##### Applying Support Vector Machine on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "71f7e6a4-680f-4307-92e7-7ef370dddf45",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Initialize the model\n",
    "SVM = SVC(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'C': [0.01, 0.1, 1, 10], # Regularization parameter\n",
    " 'kernel': ['linear', 'rbf', 'poly'], # Kernel type\n",
    " 'gamma': ['scale', 'auto'], # Kernel coefficient for 'rbf', 'poly',\n",
    " 'degree': [3, 4, 5], # Degree of the polynomial kernel funct\n",
    " 'class_weight': [None, 'balanced'] # Weighing classes in the decision func\n",
    "}\n",
    "\n",
    "\n",
    "run_my_grid_search(SVM, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"Support Vector Machine\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "a67d8c04-6720-43c1-a0f0-a83fea3c1318",
   "metadata": {},
   "source": [
    "##### Applying GaussianNB Classifier on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 117,
   "id": "a878d56f-be07-430f-a70f-ca78b3432435",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "Gaussian Naive Bayes Classifier\n",
      "Accuracy: 0.8634\n",
      "Precision: 0.8672\n",
      "Recall: 0.8634\n",
      "F1 Score: 0.8652\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'var_smoothing': 1e-06}\n"
     ]
    }
   ],
   "source": [
    "# Initialize the model\n",
    "gnb = GaussianNB()\n",
    "\n",
    "param_grid = {\n",
    " 'var_smoothing': [1e-9, 1e-8, 1e-7, 1e-6]\n",
    "}\n",
    "\n",
    "run_my_grid_search(gnb, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"Gaussian Naive Bayes Classifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "16152978-51c3-45db-b34b-0666ad280b96",
   "metadata": {},
   "source": [
    "##### Applying KNeighbors on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 118,
   "id": "fc7e192a-5744-4c20-86dd-061a775cb791",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "KNeighbors Classifier\n",
      "Accuracy: 0.8994\n",
      "Precision: 0.8937\n",
      "Recall: 0.8994\n",
      "F1 Score: 0.8844\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'algorithm': 'auto', 'metric': 'manhattan', 'n_neighbors': 9, 'weights': 'uniform'}\n"
     ]
    }
   ],
   "source": [
    "# Initialize the model\n",
    "knn = KNeighborsClassifier(n_neighbors=2)\n",
    "\n",
    "param_grid = {\n",
    " 'n_neighbors': [1, 3, 5, 7, 9], # Number of neighbors to use\n",
    " 'weights': ['uniform', 'distance'], # Weight function used in prediction\n",
    " 'metric': ['euclidean', 'manhattan'], # Distance metric\n",
    " 'algorithm': ['auto', 'ball_tree', 'kd_tree', 'brute'] # Algorithm to compute\n",
    "}\n",
    "\n",
    "run_my_grid_search(knn, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"KNeighbors Classifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "13a88bd2-8202-4db6-9c38-23ec6cf38f86",
   "metadata": {},
   "source": [
    "# ----------------------------------------"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "1e01eea3-8709-43da-8d88-6b9eccb1be8c",
   "metadata": {},
   "source": [
    "## 2. Wrapper Method (Recursive Feature Elimination - RFE)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 48,
   "id": "e2e7c83c-e9cc-46e6-bd6e-a4602346337f",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Selected Features (RFE): [ 0  2  5  7  9 12 13 14]\n"
     ]
    }
   ],
   "source": [
    "from sklearn.feature_selection import RFE\n",
    "\n",
    "clf_rf_2 = RandomForestClassifier(random_state=43)\n",
    "rfe_selector = RFE(estimator=clf_rf_2, n_features_to_select=8, step=1)\n",
    "\n",
    "X_train_selected_rfe = rfe_selector.fit_transform(X_train, y_train)\n",
    "X_test_selected_rfe = rfe_selector.transform(X_test)\n",
    "\n",
    "# Get the selected feature indices\n",
    "selected_features = rfe_selector.get_support(indices=True)\n",
    "print(\"Selected Features (RFE):\", selected_features)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "97eb6459-f178-452d-a833-e9b9e02b2ebc",
   "metadata": {},
   "source": [
    "#### Applying Random Forest on selected features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 49,
   "id": "fe341c38-b7ab-4bc8-96ec-77ca99f8177c",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.9106\n",
      "Precision: 0.9051\n",
      "Recall: 0.9106\n",
      "F1 Score: 0.9014\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': 10, 'min_samples_leaf': 2, 'min_samples_split': 5, 'n_estimators': 300}\n"
     ]
    }
   ],
   "source": [
    "# Define the Random Forest Classifier\n",
    "rf_classifier = RandomForestClassifier(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'C': [0.01, 0.1, 1, 10], # Regularization strength\n",
    " 'solver': ['liblinear', 'lbfgs'], # Solver type\n",
    " 'max_iter': [100, 200, 300] # Maximum iterations for convergence\n",
    "}\n",
    "\n",
    "\n",
    "run_my_grid_search(LR, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"LogisticRegressionClassifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "6c54569c-bd2d-4c65-bc07-3cbda949cf93",
   "metadata": {},
   "source": [
    "##### Applying Logistic Regression on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "61734869-ed06-40ec-aa0b-2438c36e4290",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Initialize the model\n",
    "LR = LogisticRegression(max_iter=5000, random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'C': [0.01, 0.1, 1, 10], # Regularization strength\n",
    " 'solver': ['liblinear', 'lbfgs'], # Solver type\n",
    " 'max_iter': [100, 200, 300] # Maximum iterations for convergence\n",
    "}\n",
    "\n",
    "run_my_grid_search(LR, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"LogisticRegressionClassifier\")"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "e197044c-64b4-4525-a1cc-25d058851209",
   "metadata": {},
   "source": [
    "##### Applying Decision Tree on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "90db8c50-4acf-4480-a5ba-31c1e8411d8a",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Initialize the model\n",
    "DT = DecisionTreeClassifier(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'criterion': ['gini', 'entropy'], # The function to measure the qual\n",
    " 'max_depth': [None, 10, 20, 30], # The maximum depth of the tree\n",
    " 'min_samples_split': [2, 5, 10], # The minimum number of samples r\n",
    " 'min_samples_leaf': [1, 2, 4], # The minimum number of samples r\n",
    " 'max_features': [None, 'sqrt', 'log2'] # The number of features to consid\n",
    "}\n",
    "\n",
    "run_my_grid_search(DT, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"DecisionTreeClassifier\")"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "dc3a916b-3d5d-4a6a-a75f-2fdc054e84e9",
   "metadata": {},
   "source": [
    "##### Applying Support Vector Machine on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "9c3b7da8-b21b-49bc-b018-cc5a40639e72",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Initialize the model\n",
    "SVM = SVC(random_state=42)\n",
    "\n",
    "# Define the hyperparameters for grid search\n",
    "param_grid = {\n",
    " 'C': [0.01, 0.1, 1, 10], # Regularization parameter\n",
    " 'kernel': ['linear', 'rbf', 'poly'], # Kernel type\n",
    " 'gamma': ['scale', 'auto'], # Kernel coefficient for 'rbf', 'poly',\n",
    " 'degree': [3, 4, 5], # Degree of the polynomial kernel funct\n",
    " 'class_weight': [None, 'balanced'] # Weighing classes in the decision func\n",
    "}\n",
    "\n",
    "\n",
    "run_my_grid_search(SVM, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"Support Vector Machine\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "9960f9ed-8c4d-4bd3-a985-f8cc04b221cf",
   "metadata": {},
   "source": [
    "##### Applying GaussianNB Classifier on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "07da0906-6d90-4bb6-9396-9f37e448c73c",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Initialize the model\n",
    "gnb = GaussianNB()\n",
    "\n",
    "param_grid = {\n",
    " 'var_smoothing': [1e-9, 1e-8, 1e-7, 1e-6]\n",
    "}\n",
    "\n",
    "run_my_grid_search(gnb, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"Gaussian Naive Bayes Classifier\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "83990ec8-12e3-4d55-b3c0-62ac335faad9",
   "metadata": {},
   "source": [
    "##### Applying KNeighbors on Selected Features"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 119,
   "id": "437fe85c-7c3a-46c0-86e2-4b630ac62fd5",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "KNeighbors Classifier\n",
      "Accuracy: 0.8994\n",
      "Precision: 0.8937\n",
      "Recall: 0.8994\n",
      "F1 Score: 0.8844\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'algorithm': 'auto', 'metric': 'manhattan', 'n_neighbors': 9, 'weights': 'uniform'}\n"
     ]
    }
   ],
   "source": [
    "# Initialize the model\n",
    "knn = KNeighborsClassifier(n_neighbors=2)\n",
    "\n",
    "param_grid = {\n",
    " 'n_neighbors': [1, 3, 5, 7, 9], # Number of neighbors to use\n",
    " 'weights': ['uniform', 'distance'], # Weight function used in prediction\n",
    " 'metric': ['euclidean', 'manhattan'], # Distance metric\n",
    " 'algorithm': ['auto', 'ball_tree', 'kd_tree', 'brute'] # Algorithm to compute\n",
    "}\n",
    "\n",
    "run_my_grid_search(knn, param_grid, X_train_selected, y_train, X_test_selected, y_test,\"KNeighbors Classifier\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "a837276a-4d9c-4dcf-9515-5d32cbdd428c",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python [conda env:base] *",
   "language": "python",
   "name": "conda-base-py"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.13.5"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
