{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "13267ec6-0689-46e5-a9cc-8e9c63a0683b",
   "metadata": {},
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "import numpy as np\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import classification_report, accuracy_score\n",
    "from sklearn.preprocessing import LabelEncoder\n",
    "from sklearn.impute import SimpleImputer\n",
    "from sklearn.feature_selection import SelectKBest, f_classif, RFE\n",
    "from sklearn.linear_model import LogisticRegression"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "0876c589-9bc6-4893-8c39-d8f3ef71d906",
   "metadata": {},
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "data = pd.read_csv(\"kidney_disease.csv\") \n",
    "data . head()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "35f555bb-4b99-4fd8-825a-ce37da650449",
   "metadata": {
    "scrolled": true
   },
   "outputs": [],
   "source": [
    "from sklearn.preprocessing import LabelEncoder\n",
    "from sklearn.impute import SimpleImputer\n",
    "df_cleaned =data.copy()\n",
    "label_encoders = {}\n",
    "for column in df_cleaned.select_dtypes(include='object').columns:\n",
    "    le = LabelEncoder()\n",
    "    df_cleaned[column] = df_cleaned[column].astype(str)\n",
    "    df_cleaned[column] = le.fit_transform(df_cleaned[column])\n",
    "    label_encoders[column] = le\n",
    "imputer = SimpleImputer(strategy='mean')\n",
    "df_cleaned = pd.DataFrame(imputer.fit_transform(df_cleaned), columns=df_cleaned.columns)\n",
    "df_cleaned.head()\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "60253b54-37ba-4c56-b789-3a5173612a40",
   "metadata": {
    "scrolled": true
   },
   "outputs": [],
   "source": [
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import classification_report, accuracy_score\n",
    "\n",
    "X = df_cleaned.drop(columns=[ 'classification'])\n",
    "y = df_cleaned['classification']\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n",
    "\n",
    "param_grid = {\n",
    "    'n_estimators': [50, 100],\n",
    "    'max_depth': [5, 10, None]\n",
    "}\n",
    "\n",
    "grid_search = GridSearchCV(RandomForestClassifier(random_state=42), param_grid, cv=3)\n",
    "grid_search.fit(X_train, y_train)\n",
    "\n",
    "best_model = grid_search.best_estimator_\n",
    "y_pred = best_model.predict(X_test)\n",
    "\n",
    "print(\"Best Parameters:\", grid_search.best_params_)\n",
    "print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n",
    "print(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "4d061b8e-6328-4c68-a34a-31fb25efd4d9",
   "metadata": {
    "scrolled": true
   },
   "outputs": [],
   "source": [
    "param_grid = {\n",
    "    'n_estimators': [50, 100],\n",
    "    'max_depth': [5, 10, None]\n",
    "}\n",
    "grid_search = GridSearchCV(RandomForestClassifier(random_state=42), param_grid, cv=3)\n",
    "grid_search.fit(X_train, y_train)\n",
    "\n",
    "best_model = grid_search.best_estimator_\n",
    "y_pred = best_model.predict(X_test)\n",
    "print(classification_report(y_test, y_pred))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "a4e32676-a641-4a5d-ba0a-067c98c60f88",
   "metadata": {},
   "outputs": [],
   "source": [
    " X_selected = X[selected_kbest_features]\n",
    "X_train_sel, X_test_sel, y_train_sel, y_test_sel = train_test_split(X_selected, y, test_size=0.2, random_state=42)\n",
    "model_sel = RandomForestClassifier(random_state=42)\n",
    "model_sel.fit(X_train_sel, y_train_sel)\n",
    "y_pred_sel = model_sel.predict(X_test_sel)\n",
    "print(\"Accuracy with selected features:\", accuracy_score(y_test_sel, y_pred_sel))\n",
    "print(\"Classification Report:\\n\", classification_report(y_test_sel, y_pred_sel))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "50453d50-7146-401d-b6e0-307ceb016aec",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_kbest_selected = X[selected_kbest_features]\n",
    "X_train_kbest, X_test_kbest, y_train_kbest, y_test_kbest = train_test_split(X_kbest_selected, y, test_size=0.2, random_state=42)\n",
    "\n",
    "model_kbest = RandomForestClassifier(random_state=42)\n",
    "model_kbest.fit(X_train_kbest, y_train_kbest)\n",
    "y_pred_kbest = model_kbest.predict(X_test_kbest)\n",
    "print(classification_report(y_test_kbest, y_pred_kbest))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "1f47fa31-897b-477a-951f-ad9eec1bb1c3",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3 (ipykernel)",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.7"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
