{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "79de42c6-9320-42a0-81ea-8dc92bf35d02",
   "metadata": {},
   "outputs": [],
   "source": [
    "\n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.impute import SimpleImputer\n",
    "from sklearn.preprocessing import StandardScaler\n",
    "from sklearn.pipeline import Pipeline\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\n",
    "from sklearn.feature_selection import SelectKBest, mutual_info_classif, RFE\n",
    "from sklearn.metrics import classification_report"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "bd84cbb3-7a0c-4e55-ab2d-e179100a8c4f",
   "metadata": {},
   "outputs": [],
   "source": [
    "diabetes_data = pd.read_csv('./datasets/diabetes.csv')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "90dd803f-cbb6-4708-96f1-2ce66c00195b",
   "metadata": {},
   "outputs": [],
   "source": [
    "imputer = SimpleImputer(strategy='mean')\n",
    "X = diabetes_data.drop('Outcome', axis=1)\n",
    "y = diabetes_data['Outcome']\n",
    "X_imputed = pd.DataFrame(imputer.fit_transform(X), columns=X.columns)\n",
    "\n",
    "# تقسيم البيانات\n",
    "X_train, X_test, y_train, y_test = train_test_split(\n",
    "    X_imputed, y, test_size=0.2, random_state=42, stratify=y\n",
    ")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "id": "ab02a84f-4ad5-4e62-9f07-a88157b2b99c",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Top features (Mutual Information): ['Pregnancies', 'Glucose', 'Insulin', 'BMI', 'Age']\n",
      "Top features (RFE): ['Pregnancies', 'Glucose', 'BMI', 'DiabetesPedigreeFunction', 'Age']\n"
     ]
    }
   ],
   "source": [
    "# --- 3a: Mutual Information ---\n",
    "k_mi = max(5, int(0.5 * X_train.shape[1]))\n",
    "selector_mi = SelectKBest(score_func=mutual_info_classif, k=k_mi)\n",
    "selector_mi.fit(X_train, y_train)\n",
    "mi_features = X_train.columns[selector_mi.get_support()].tolist()\n",
    "print(\"Top features (Mutual Information):\", mi_features)\n",
    "\n",
    "# --- 3b: RFE ---\n",
    "rfe_model = LogisticRegression(max_iter=2000)\n",
    "k_rfe = max(5, int(0.5 * X_train.shape[1]))\n",
    "selector_rfe = RFE(rfe_model, n_features_to_select=k_rfe)\n",
    "selector_rfe.fit(X_train, y_train)\n",
    "rfe_features = X_train.columns[selector_rfe.support_].tolist()\n",
    "print(\"Top features (RFE):\", rfe_features)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "3a86407f-76c7-4c68-820a-d3da19721e27",
   "metadata": {},
   "outputs": [],
   "source": [
    "models = {\n",
    "    'LogisticRegression': LogisticRegression(max_iter=2000),\n",
    "    'RandomForest': RandomForestClassifier(random_state=42),\n",
    "    'GradientBoosting': GradientBoostingClassifier(random_state=42)\n",
    "}\n",
    "\n",
    "param_grids = {\n",
    "    'LogisticRegression': {'clf__C': [0.1, 1.0, 10.0]},\n",
    "    'RandomForest': {'clf__n_estimators': [50, 100, 200],\n",
    "                     'clf__max_depth': [5, 10, None]},\n",
    "    'GradientBoosting': {'clf__n_estimators': [50, 100, 200],\n",
    "                         'clf__learning_rate': [0.01, 0.1, 0.2],\n",
    "                         'clf__max_depth': [3, 5, 7]}\n",
    "}\n",
    "\n",
    "def train_grid(models, param_grids, X_train, X_test, y_train, y_test):\n",
    "    results = {}\n",
    "    for name, model in models.items():\n",
    "        pipe = Pipeline([\n",
    "            ('scaler', StandardScaler()),\n",
    "            ('clf', model)\n",
    "        ])\n",
    "        grid = GridSearchCV(pipe, param_grids[name], cv=5, scoring='accuracy', n_jobs=-1)\n",
    "        grid.fit(X_train, y_train)\n",
    "        y_pred = grid.predict(X_test)\n",
    "        print(f\"\\n--- {name} ---\")\n",
    "        print(f\"Best Parameters: {grid.best_params_}\")\n",
    "        print(classification_report(y_test, y_pred))\n",
    "        results[name] = grid\n",
    "    return results"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "id": "6f1a7aec-9e92-4ea2-bb1d-44be68bdd157",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "### Models on all features ###\n",
      "\n",
      "--- LogisticRegression ---\n",
      "Best Parameters: {'clf__C': 1.0}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.76      0.82      0.79       100\n",
      "           1       0.61      0.52      0.56        54\n",
      "\n",
      "    accuracy                           0.71       154\n",
      "   macro avg       0.68      0.67      0.67       154\n",
      "weighted avg       0.71      0.71      0.71       154\n",
      "\n",
      "\n",
      "--- RandomForest ---\n",
      "Best Parameters: {'clf__max_depth': 10, 'clf__n_estimators': 100}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.80      0.83      0.81       100\n",
      "           1       0.66      0.61      0.63        54\n",
      "\n",
      "    accuracy                           0.75       154\n",
      "   macro avg       0.73      0.72      0.72       154\n",
      "weighted avg       0.75      0.75      0.75       154\n",
      "\n",
      "\n",
      "--- GradientBoosting ---\n",
      "Best Parameters: {'clf__learning_rate': 0.01, 'clf__max_depth': 3, 'clf__n_estimators': 200}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.78      0.85      0.81       100\n",
      "           1       0.67      0.56      0.61        54\n",
      "\n",
      "    accuracy                           0.75       154\n",
      "   macro avg       0.72      0.70      0.71       154\n",
      "weighted avg       0.74      0.75      0.74       154\n",
      "\n"
     ]
    }
   ],
   "source": [
    "print(\"\\n### Models on all features ###\")\n",
    "results_all = train_grid(models, param_grids, X_train, X_test, y_train, y_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "958fd791-bb19-4329-8dc1-43921124a819",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "### Models on MI features ###\n",
      "\n",
      "--- LogisticRegression ---\n",
      "Best Parameters: {'clf__C': 0.1}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.76      0.82      0.79       100\n",
      "           1       0.61      0.52      0.56        54\n",
      "\n",
      "    accuracy                           0.71       154\n",
      "   macro avg       0.68      0.67      0.67       154\n",
      "weighted avg       0.71      0.71      0.71       154\n",
      "\n",
      "\n",
      "--- RandomForest ---\n",
      "Best Parameters: {'clf__max_depth': 5, 'clf__n_estimators': 200}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.78      0.83      0.80       100\n",
      "           1       0.64      0.56      0.59        54\n",
      "\n",
      "    accuracy                           0.73       154\n",
      "   macro avg       0.71      0.69      0.70       154\n",
      "weighted avg       0.73      0.73      0.73       154\n",
      "\n",
      "\n",
      "--- GradientBoosting ---\n",
      "Best Parameters: {'clf__learning_rate': 0.1, 'clf__max_depth': 3, 'clf__n_estimators': 50}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.79      0.84      0.82       100\n",
      "           1       0.67      0.59      0.63        54\n",
      "\n",
      "    accuracy                           0.75       154\n",
      "   macro avg       0.73      0.72      0.72       154\n",
      "weighted avg       0.75      0.75      0.75       154\n",
      "\n"
     ]
    }
   ],
   "source": [
    "print(\"\\n### Models on MI features ###\")\n",
    "X_train_mi = X_train[mi_features]\n",
    "X_test_mi  = X_test[mi_features]\n",
    "results_mi = train_grid(models, param_grids, X_train_mi, X_test_mi, y_train, y_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "id": "e58e7f92-c3dd-4302-a5f5-63d93dc61a5d",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "### Models on RFE features ###\n",
      "\n",
      "--- LogisticRegression ---\n",
      "Best Parameters: {'clf__C': 10.0}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.75      0.81      0.78       100\n",
      "           1       0.59      0.50      0.54        54\n",
      "\n",
      "    accuracy                           0.70       154\n",
      "   macro avg       0.67      0.66      0.66       154\n",
      "weighted avg       0.69      0.70      0.70       154\n",
      "\n",
      "\n",
      "--- RandomForest ---\n",
      "Best Parameters: {'clf__max_depth': 5, 'clf__n_estimators': 100}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.78      0.87      0.82       100\n",
      "           1       0.69      0.54      0.60        54\n",
      "\n",
      "    accuracy                           0.75       154\n",
      "   macro avg       0.73      0.70      0.71       154\n",
      "weighted avg       0.75      0.75      0.74       154\n",
      "\n",
      "\n",
      "--- GradientBoosting ---\n",
      "Best Parameters: {'clf__learning_rate': 0.01, 'clf__max_depth': 3, 'clf__n_estimators': 200}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "           0       0.78      0.85      0.81       100\n",
      "           1       0.67      0.56      0.61        54\n",
      "\n",
      "    accuracy                           0.75       154\n",
      "   macro avg       0.72      0.70      0.71       154\n",
      "weighted avg       0.74      0.75      0.74       154\n",
      "\n"
     ]
    }
   ],
   "source": [
    "print(\"\\n### Models on RFE features ###\")\n",
    "X_train_rfe = X_train[rfe_features]\n",
    "X_test_rfe  = X_test[rfe_features]\n",
    "results_rfe = train_grid(models, param_grids, X_train_rfe, X_test_rfe, y_train, y_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "id": "637d890e-085c-4a57-a85f-e5ad4e53cace",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "File saved as BashayerShaheen.py\n"
     ]
    }
   ],
   "source": [
    "file_name = \"BashayerShaheen.py\"\n",
    "\n",
    "code_string = \"\"\"\n",
    "\"\"\"\n",
    "with open(file_name, \"w\") as f:\n",
    "    f.write(code_string)\n",
    "print(f\"File saved as {file_name}\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "f81da3b9-2349-4a18-b48e-048bbe787c57",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python [conda env:base] *",
   "language": "python",
   "name": "conda-base-py"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.13.5"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
