{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "d0fc441d-a873-4a73-8afb-5643f99ca4c3",
   "metadata": {
    "scrolled": true
   },
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "import numpy as np\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.preprocessing import StandardScaler\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import classification_report\n",
    "from sklearn.feature_selection import SelectKBest, f_classif\n",
    "import warnings\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 14,
   "id": "d995d276-13dc-4cc1-83e9-d89295ad2d08",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 1: Dataset Loading and Cleaning ---\n",
      "Dataset loaded successfully.\n",
      "Cleaned Column Names:\n",
      "Index(['age', 'sex', 'cp', 'trestbps', 'chol', 'fbs', 'restecg', 'thalach',\n",
      "       'exang', 'oldpeak', 'slope', 'ca', 'thal', 'num'],\n",
      "      dtype='object')\n"
     ]
    }
   ],
   "source": [
    "print(\"--- Step 1: Dataset Loading and Cleaning ---\")\n",
    "try:\n",
    "    df = pd.read_csv('Heart Attack Prediction.csv')\n",
    "    df.columns = df.columns.str.strip()\n",
    "    print(\"Dataset loaded successfully.\")\n",
    "    print(\"Cleaned Column Names:\")\n",
    "    print(df.columns)\n",
    "except FileNotFoundError:\n",
    "    print(\"Error: 'Heart Attack Prediction.csv' file not found.\")\n",
    "    exit()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "id": "937e6615-6661-4dff-ab16-553806b16115",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 2: Handling Missing Values ---\n",
      "Missing values handled.\n",
      "Missing values after imputation:\n",
      "age         0\n",
      "sex         0\n",
      "cp          0\n",
      "trestbps    0\n",
      "chol        0\n",
      "fbs         0\n",
      "restecg     0\n",
      "thalach     0\n",
      "exang       0\n",
      "oldpeak     0\n",
      "slope       0\n",
      "ca          0\n",
      "thal        0\n",
      "num         0\n",
      "dtype: int64\n"
     ]
    }
   ],
   "source": [
    "# Step 2: Handle Missing Values\n",
    "print(\"--- Step 2: Handling Missing Values ---\")\n",
    "df['ca'] = df['ca'].replace('?', np.nan)\n",
    "df['thal'] = df['thal'].replace('?', np.nan)\n",
    "df['ca'] = pd.to_numeric(df['ca'])\n",
    "df['thal'] = pd.to_numeric(df['thal'])\n",
    "df['ca'] = df['ca'].replace(4, np.nan)\n",
    "df['thal'] = df['thal'].replace(0, np.nan)\n",
    "for col in ['ca', 'thal']:\n",
    "    df[col] = df[col].fillna(df[col].mode()[0])\n",
    "print(\"Missing values handled.\")\n",
    "print(\"Missing values after imputation:\")\n",
    "print(df.isnull().sum())"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 16,
   "id": "b366e7c8-9ecb-456b-910c-d60e81df4982",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 3: Applying ML Models with Grid Search ---\n",
      "Data split and scaled.\n"
     ]
    }
   ],
   "source": [
    "# Step 3: Apply ML Models with Grid Search\n",
    "print(\"--- Step 3: Applying ML Models with Grid Search ---\")\n",
    "target_column_name = \"num\"\n",
    "try:\n",
    "    X = df.drop(target_column_name, axis=1)\n",
    "    y = df[target_column_name]\n",
    "except KeyError:\n",
    "    print(f\"Error: The target column '{target_column_name}' was not found. Please re-check the column names.\")\n",
    "    exit()\n",
    "\n",
    "for col in X.columns:\n",
    "    X[col] = pd.to_numeric(X[col], errors='coerce')\n",
    "X = X.dropna(axis=1, how='any')\n",
    "X = X.dropna()\n",
    "y = y[X.index]\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n",
    "scaler = StandardScaler()\n",
    "X_train_scaled = scaler.fit_transform(X_train)\n",
    "X_test_scaled = scaler.transform(X_test)\n",
    "print(\"Data split and scaled.\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 17,
   "id": "01a809ff-d653-4728-bd1c-08cd58db8bac",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "--- Logistic Regression Grid Search ---\n",
      "Logistic Regression Grid Search completed.\n",
      "Best parameters: {'C': 0.001, 'solver': 'liblinear'}\n",
      "\n",
      "Classification Report on Test Data (Logistic Regression):\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.85      0.76      0.81        38\n",
      "           1       0.64      0.76      0.70        21\n",
      "\n",
      "    accuracy                           0.76        59\n",
      "   macro avg       0.75      0.76      0.75        59\n",
      "weighted avg       0.78      0.76      0.77        59\n",
      "\n"
     ]
    }
   ],
   "source": [
    "# Logistic Regression\n",
    "print(\"\\n--- Logistic Regression Grid Search ---\")\n",
    "param_grid_lr = {'C': [0.001, 0.01, 0.1, 1, 10, 100], 'solver': ['liblinear']}\n",
    "grid_search_lr = GridSearchCV(LogisticRegression(max_iter=1000), param_grid_lr, cv=5)\n",
    "with warnings.catch_warnings():\n",
    "    warnings.filterwarnings('ignore', category=UserWarning)\n",
    "    grid_search_lr.fit(X_train_scaled, y_train)\n",
    "print(\"Logistic Regression Grid Search completed.\")\n",
    "print(\"Best parameters:\", grid_search_lr.best_params_)\n",
    "y_pred_lr = grid_search_lr.predict(X_test_scaled)\n",
    "print(\"\\nClassification Report on Test Data (Logistic Regression):\")\n",
    "print(classification_report(y_test, y_pred_lr))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 18,
   "id": "f6c2e4e0-4f5a-459d-955f-5acaaeafd5a0",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "--- Random Forest Grid Search ---\n",
      "Random Forest Grid Search completed.\n",
      "Best parameters: {'max_depth': 5, 'n_estimators': 200}\n",
      "\n",
      "Classification Report on Test Data (Random Forest):\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.83      0.79      0.81        38\n",
      "           1       0.65      0.71      0.68        21\n",
      "\n",
      "    accuracy                           0.76        59\n",
      "   macro avg       0.74      0.75      0.75        59\n",
      "weighted avg       0.77      0.76      0.76        59\n",
      "\n"
     ]
    }
   ],
   "source": [
    "# Random Forest\n",
    "print(\"\\n--- Random Forest Grid Search ---\")\n",
    "param_grid_rf = {'n_estimators': [50, 100, 200], 'max_depth': [5, 10, 15]}\n",
    "grid_search_rf = GridSearchCV(RandomForestClassifier(random_state=42), param_grid_rf, cv=5)\n",
    "grid_search_rf.fit(X_train, y_train)\n",
    "print(\"Random Forest Grid Search completed.\")\n",
    "print(\"Best parameters:\", grid_search_rf.best_params_)\n",
    "y_pred_rf = grid_search_rf.predict(X_test)\n",
    "print(\"\\nClassification Report on Test Data (Random Forest):\")\n",
    "print(classification_report(y_test, y_pred_rf))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 19,
   "id": "1db39c9b-9431-4be8-9b4e-38fb150d9c8d",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 4: Performing Feature Selection ---\n",
      "Feature importance from Random Forest calculated.\n",
      "\n",
      "Feature Importance from Random Forest:\n",
      "   feature  importance\n",
      "3  oldpeak    0.389775\n",
      "2       cp    0.352867\n",
      "0      age    0.139262\n",
      "1      sex    0.090855\n",
      "5     thal    0.027241\n",
      "4       ca    0.000000\n",
      "\n",
      "Warning: The following features have constant values and will be removed: ['ca']\n",
      "\n",
      "Top 5 features selected by SelectKBest.\n",
      "Top 5 Features selected by SelectKBest:\n",
      "['age', 'sex', 'cp', 'oldpeak', 'thal']\n"
     ]
    }
   ],
   "source": [
    "# Step 4: Perform Feature Selection\n",
    "print(\"--- Step 4: Performing Feature Selection ---\")\n",
    "importances = grid_search_rf.best_estimator_.feature_importances_\n",
    "feature_names = X.columns\n",
    "feature_importance_df = pd.DataFrame({'feature': feature_names, 'importance': importances}).sort_values(by='importance', ascending=False)\n",
    "print(\"Feature importance from Random Forest calculated.\")\n",
    "print(\"\\nFeature Importance from Random Forest:\")\n",
    "print(feature_importance_df)\n",
    "k = 5\n",
    "constant_features = [col for col in X.columns if X[col].nunique() == 1]\n",
    "if constant_features:\n",
    "    print(f\"\\nWarning: The following features have constant values and will be removed: {constant_features}\")\n",
    "    X_filtered = X.drop(columns=constant_features)\n",
    "else:\n",
    "    X_filtered = X.copy()\n",
    "selector = SelectKBest(f_classif, k=k)\n",
    "selector.fit(X_filtered, y)\n",
    "selected_features_mask = selector.get_support()\n",
    "selected_features_kbest = X_filtered.columns[selected_features_mask]\n",
    "print(f\"\\nTop {k} features selected by SelectKBest.\")\n",
    "print(f\"Top {k} Features selected by SelectKBest:\")\n",
    "print(selected_features_kbest.tolist())"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 20,
   "id": "100cf5b5-59e9-4f5e-8ff3-531097f68a2b",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 5: Applying Models to Selected Features ---\n",
      "Data split and scaled with selected features.\n"
     ]
    }
   ],
   "source": [
    "# Step 5: Apply Models to Selected Features\n",
    "print(\"--- Step 5: Applying Models to Selected Features ---\")\n",
    "X_selected = X[selected_features_kbest]\n",
    "X_train_sel, X_test_sel, y_train_sel, y_test_sel = train_test_split(X_selected, y, test_size=0.2, random_state=42)\n",
    "scaler_sel = StandardScaler()\n",
    "X_train_sel_scaled = scaler_sel.fit_transform(X_train_sel)\n",
    "X_test_sel_scaled = scaler_sel.transform(X_test_sel)\n",
    "print(\"Data split and scaled with selected features.\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 21,
   "id": "e743baa6-eec2-40de-978b-c74fff2b7422",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "--- Logistic Regression with Selected Features ---\n",
      "Model trained and predictions made.\n",
      "Classification Report:\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.85      0.76      0.81        38\n",
      "           1       0.64      0.76      0.70        21\n",
      "\n",
      "    accuracy                           0.76        59\n",
      "   macro avg       0.75      0.76      0.75        59\n",
      "weighted avg       0.78      0.76      0.77        59\n",
      "\n"
     ]
    }
   ],
   "source": [
    "# Logistic Regression with selected features\n",
    "lr_selected = LogisticRegression(**grid_search_lr.best_params_)\n",
    "lr_selected.fit(X_train_sel_scaled, y_train_sel)\n",
    "y_pred_lr_sel = lr_selected.predict(X_test_sel_scaled)\n",
    "print(\"\\n--- Logistic Regression with Selected Features ---\")\n",
    "print(\"Model trained and predictions made.\")\n",
    "print(\"Classification Report:\")\n",
    "print(classification_report(y_test_sel, y_pred_lr_sel))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 22,
   "id": "ee99ce5b-1f6f-4fae-ac2c-0c0939e4d483",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "--- Random Forest with Selected Features ---\n",
      "Model trained and predictions made.\n",
      "Classification Report:\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.84      0.82      0.83        38\n",
      "           1       0.68      0.71      0.70        21\n",
      "\n",
      "    accuracy                           0.78        59\n",
      "   macro avg       0.76      0.77      0.76        59\n",
      "weighted avg       0.78      0.78      0.78        59\n",
      "\n"
     ]
    }
   ],
   "source": [
    "# Random Forest with selected features\n",
    "rf_selected = RandomForestClassifier(**grid_search_rf.best_params_, random_state=42)\n",
    "rf_selected.fit(X_train_sel, y_train_sel)\n",
    "y_pred_rf_sel = rf_selected.predict(X_test_sel)\n",
    "print(\"\\n--- Random Forest with Selected Features ---\")\n",
    "print(\"Model trained and predictions made.\")\n",
    "print(\"Classification Report:\")\n",
    "print(classification_report(y_test_sel, y_pred_rf_sel))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 23,
   "id": "7acd8b1b-923e-431b-a368-8f682563473d",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "--- Step 6: Saving and Final Review ---\n",
      "Processed dataset saved to: Heart_Attack_Processed_Dataset.csv\n",
      "\n",
      "Final DataFrame Info after all operations:\n",
      "<class 'pandas.core.frame.DataFrame'>\n",
      "RangeIndex: 294 entries, 0 to 293\n",
      "Data columns (total 14 columns):\n",
      " #   Column    Non-Null Count  Dtype  \n",
      "---  ------    --------------  -----  \n",
      " 0   age       294 non-null    int64  \n",
      " 1   sex       294 non-null    int64  \n",
      " 2   cp        294 non-null    int64  \n",
      " 3   trestbps  294 non-null    object \n",
      " 4   chol      294 non-null    object \n",
      " 5   fbs       294 non-null    object \n",
      " 6   restecg   294 non-null    object \n",
      " 7   thalach   294 non-null    object \n",
      " 8   exang     294 non-null    object \n",
      " 9   oldpeak   294 non-null    float64\n",
      " 10  slope     294 non-null    object \n",
      " 11  ca        294 non-null    float64\n",
      " 12  thal      294 non-null    float64\n",
      " 13  num       294 non-null    int64  \n",
      "dtypes: float64(3), int64(4), object(7)\n",
      "memory usage: 32.3+ KB\n",
      "\n",
      "First 5 rows of the processed DataFrame:\n",
      "   age  sex  cp trestbps chol fbs restecg thalach exang  oldpeak slope   ca  \\\n",
      "0   28    1   2      130  132   0       2     185     0      0.0     ?  0.0   \n",
      "1   29    1   2      120  243   0       0     160     0      0.0     ?  0.0   \n",
      "2   29    1   2      140    ?   0       0     170     0      0.0     ?  0.0   \n",
      "3   30    0   1      170  237   0       1     170     0      0.0     ?  0.0   \n",
      "4   31    0   2      100  219   0       1     150     0      0.0     ?  0.0   \n",
      "\n",
      "   thal  num  \n",
      "0   7.0    0  \n",
      "1   7.0    0  \n",
      "2   7.0    0  \n",
      "3   6.0    0  \n",
      "4   7.0    0  \n"
     ]
    }
   ],
   "source": [
    "# Step 6: Final Output\n",
    "print(\"--- Step 6: Saving and Final Review ---\")\n",
    "output_file_name = 'Heart_Attack_Processed_Dataset.csv'\n",
    "df.to_csv(output_file_name, index=False)\n",
    "print(f\"Processed dataset saved to: {output_file_name}\")\n",
    "\n",
    "print(\"\\nFinal DataFrame Info after all operations:\")\n",
    "df.info()\n",
    "\n",
    "print(\"\\nFirst 5 rows of the processed DataFrame:\")\n",
    "print(df.head())"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "b726273a-a35d-4910-9f7b-49d445ce5700",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python [conda env:base] *",
   "language": "python",
   "name": "conda-base-py"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.7"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
