{
  "nbformat": 4,
  "nbformat_minor": 5,
  "metadata": {
    "colab": {
      "name": "CKD_Full_Pipeline_Lama_AlSubaie.ipynb"
    },
    "kernelspec": {
      "name": "python3",
      "display_name": "Python 3"
    },
    "language_info": {
      "name": "python"
    }
  },
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {},
      "source": [
        "# Chronic Kidney Disease (CKD) \u2014 Full ML Pipeline\n",
        "Author: **Lama Mansour AlSubaie**\n",
        "\n",
        "This notebook builds an end-to-end pipeline for a cleaned CKD dataset:\n",
        "- EDA\n",
        "- Train/Test split\n",
        "- Models: Logistic Regression, SVM, Random Forest (with GridSearchCV)\n",
        "- Feature selection: SelectKBest (Mutual Information) & RFE (L1-LogReg)\n",
        "- Retrain & evaluate on selected features\n",
        "- Save artifacts (CSV/PNG/JOBLIB/JSON) and a ZIP for easy download\n",
        "\n",
        "> **Note:** Please upload your file `kidney_disease_clean_final.csv` when prompted."
      ]
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "setup"
      },
      "source": [
        "# If running in Google Colab, uncomment to ensure dependencies are present\n",
        "# !pip install -q scikit-learn matplotlib numpy pandas joblib"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "upload"
      },
      "source": [
        "import os, json, numpy as np, pandas as pd, matplotlib.pyplot as plt\n",
        "from io import BytesIO\n",
        "try:\n",
        "    from google.colab import files  # type: ignore\n",
        "    print('>> Please upload kidney_disease_clean_final.csv')\n",
        "    uploaded = files.upload()\n",
        "except Exception as e:\n",
        "    print('Not in Colab or files.upload() unavailable. Make sure the CSV is in the current directory.')\n",
        "\n",
        "DATA_PATH = 'kidney_disease_clean_final.csv'\n",
        "assert os.path.exists(DATA_PATH), 'CSV file not found. Ensure kidney_disease_clean_final.csv is present.'\n",
        "df = pd.read_csv(DATA_PATH)\n",
        "print(df.shape)\n",
        "df.head(3)"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "eda"
      },
      "source": [
        "# ==================== EDA ====================\n",
        "eda = {\n",
        "    'rows': int(df.shape[0]),\n",
        "    'cols': int(df.shape[1]),\n",
        "    'columns': df.columns.tolist(),\n",
        "    'dtypes': df.dtypes.astype(str).to_dict(),\n",
        "    'class_balance': df['classification'].value_counts().to_dict(),\n",
        "    'missing_counts': df.isna().sum().to_dict()\n",
        "}\n",
        "with open('EDA_summary.json', 'w') as f:\n",
        "    json.dump(eda, f, indent=2)\n",
        "pd.DataFrame([{\n",
        "    'rows': eda['rows'], 'cols': eda['cols'], 'classes': eda['class_balance']\n",
        "}])"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "split_cv"
      },
      "source": [
        "# ================= Train/Test Split & CV =================\n",
        "from sklearn.model_selection import train_test_split, StratifiedKFold\n",
        "y = df['classification'].astype(int).values\n",
        "X = df.drop(columns=['classification'])\n",
        "feature_names = X.columns.tolist()\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, stratify=y, random_state=42)\n",
        "cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n",
        "X_train.shape, X_test.shape"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "models_grids"
      },
      "source": [
        "# ================= Models & Grids =================\n",
        "from sklearn.preprocessing import StandardScaler\n",
        "from sklearn.pipeline import Pipeline\n",
        "from sklearn.linear_model import LogisticRegression\n",
        "from sklearn.svm import SVC\n",
        "from sklearn.ensemble import RandomForestClassifier\n",
        "from sklearn.model_selection import GridSearchCV\n",
        "from sklearn.metrics import (accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, RocCurveDisplay)\n",
        "\n",
        "pipelines_and_grids = {\n",
        "    'LogisticRegression': (\n",
        "        Pipeline([('scaler', StandardScaler()), ('clf', LogisticRegression(max_iter=2000, solver='liblinear'))]),\n",
        "        { 'clf__penalty': ['l1', 'l2'], 'clf__C': [0.01, 0.1, 1, 10, 100] }\n",
        "    ),\n",
        "    'SVM': (\n",
        "        Pipeline([('scaler', StandardScaler()), ('clf', SVC(probability=True))]),\n",
        "        { 'clf__kernel': ['rbf', 'linear'], 'clf__C': [0.1, 1, 10, 100], 'clf__gamma': ['scale', 'auto'] }\n",
        "    ),\n",
        "    'RandomForest': (\n",
        "        Pipeline([('clf', RandomForestClassifier(random_state=42))]),\n",
        "        { 'clf__n_estimators': [200, 400], 'clf__max_depth': [None, 5, 10, 20], 'clf__min_samples_split': [2, 5], 'clf__min_samples_leaf': [1, 2] }\n",
        "    )\n",
        "}\n",
        "\n",
        "results_rows = []\n",
        "best_ests = {}\n",
        "for name, (pipe, grid) in pipelines_and_grids.items():\n",
        "    gs = GridSearchCV(pipe, grid, scoring='roc_auc', cv=cv, n_jobs=-1, refit=True)\n",
        "    gs.fit(X_train, y_train)\n",
        "    y_proba = gs.best_estimator_.predict_proba(X_test)[:, 1]\n",
        "    y_pred = gs.best_estimator_.predict(X_test)\n",
        "    results_rows.append({\n",
        "        'model': name,\n",
        "        'best_params': str(gs.best_params_),\n",
        "        'cv_best_auc': round(gs.best_score_, 4),\n",
        "        'test_accuracy': round(accuracy_score(y_test, y_pred), 4),\n",
        "        'test_precision': round(precision_score(y_test, y_pred), 4),\n",
        "        'test_recall': round(recall_score(y_test, y_pred), 4),\n",
        "        'test_f1': round(f1_score(y_test, y_pred), 4),\n",
        "        'test_roc_auc': round(roc_auc_score(y_test, y_proba), 4)\n",
        "    })\n",
        "    best_ests[name] = gs.best_estimator_\n",
        "\n",
        "import pandas as pd\n",
        "results_df = pd.DataFrame(results_rows).sort_values('test_roc_auc', ascending=False)\n",
        "results_df.to_csv('model_results.csv', index=False)\n",
        "results_df"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "roc_plot"
      },
      "source": [
        "# ================= ROC Curves =================\n",
        "import matplotlib.pyplot as plt\n",
        "plt.figure()\n",
        "for name, est in best_ests.items():\n",
        "    RocCurveDisplay.from_estimator(est, X_test, y_test)\n",
        "plt.title('ROC Curves - All Features')\n",
        "plt.savefig('roc_curves_all_features.png', bbox_inches='tight')\n",
        "plt.close()\n",
        "print('Saved roc_curves_all_features.png')"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "selectkbest"
      },
      "source": [
        "# ================= Feature Selection A: SelectKBest (Mutual Information) =================\n",
        "from sklearn.feature_selection import SelectKBest, mutual_info_classif\n",
        "max_k = min(15, X_train.shape[1])\n",
        "k_values = list(range(5, max_k+1))\n",
        "k_search = []\n",
        "\n",
        "for k in k_values:\n",
        "    skb = SelectKBest(score_func=mutual_info_classif, k=k)\n",
        "    pipe = Pipeline([('selector', skb), ('scaler', StandardScaler()), ('clf', LogisticRegression(max_iter=2000, solver='liblinear'))])\n",
        "    params = {'clf__penalty': ['l1','l2'], 'clf__C':[0.1, 1, 10]}\n",
        "    gs = GridSearchCV(pipe, params, scoring='roc_auc', cv=cv, n_jobs=-1, refit=True)\n",
        "    gs.fit(X_train, y_train)\n",
        "    proba = gs.best_estimator_.predict_proba(X_test)[:,1]\n",
        "    pred  = gs.best_estimator_.predict(X_test)\n",
        "    from sklearn.metrics import roc_auc_score, f1_score\n",
        "    k_search.append({'k': k, 'cv_best_auc': round(gs.best_score_,4), 'test_auc': round(roc_auc_score(y_test, proba),4), 'test_f1': round(f1_score(y_test, pred),4)})\n",
        "\n",
        "k_search_df = pd.DataFrame(k_search).sort_values('test_auc', ascending=False)\n",
        "k_search_df.to_csv('selectkbest_search.csv', index=False)\n",
        "k_search_df.head()"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "mi_scores_plot"
      },
      "source": [
        "# Save MI scores and plot for top selected features\n",
        "mi_scores = mutual_info_classif(X_train, y_train, random_state=42)\n",
        "mi_df = pd.DataFrame({'feature': feature_names, 'mutual_info': mi_scores}).sort_values('mutual_info', ascending=False)\n",
        "mi_df.to_csv('feature_scores_mutual_info.csv', index=False)\n",
        "\n",
        "best_k = int(k_search_df.iloc[0]['k'])\n",
        "skb_best = SelectKBest(mutual_info_classif, k=best_k).fit(X_train, y_train)\n",
        "selected_features_skb = np.array(feature_names)[skb_best.get_support()].tolist()\n",
        "\n",
        "plt.figure()\n",
        "mi_df.set_index('feature').loc[selected_features_skb].sort_values('mutual_info', ascending=True).plot(kind='barh', legend=False)\n",
        "plt.title('Top Features by Mutual Information (Selected)')\n",
        "plt.xlabel('Mutual Information')\n",
        "plt.tight_layout()\n",
        "plt.savefig('feature_importance_mutual_info.png', bbox_inches='tight')\n",
        "plt.close()\n",
        "print('Saved feature_scores_mutual_info.csv and feature_importance_mutual_info.png')"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "rfe"
      },
      "source": [
        "# ================= Feature Selection B: RFE with L1-Logistic Regression =================\n",
        "from sklearn.feature_selection import RFE\n",
        "base_lr = LogisticRegression(max_iter=2000, solver='liblinear', penalty='l1', C=1.0)\n",
        "scaler = StandardScaler()\n",
        "X_train_scaled = scaler.fit_transform(X_train)\n",
        "rfe = RFE(estimator=base_lr, n_features_to_select=best_k, step=1)\n",
        "rfe.fit(X_train_scaled, y_train)\n",
        "rfe_support = rfe.get_support()\n",
        "selected_features_rfe = np.array(feature_names)[rfe_support].tolist()\n",
        "rfe_df = pd.DataFrame({'feature': feature_names, 'rfe_rank': rfe.ranking_}).sort_values('rfe_rank')\n",
        "rfe_df.to_csv('feature_ranks_rfe.csv', index=False)\n",
        "\n",
        "plt.figure()\n",
        "rfe_df[rfe_df['feature'].isin(selected_features_rfe)].set_index('feature').sort_values('rfe_rank', ascending=False).plot(kind='barh', legend=False)\n",
        "plt.title('Selected Features by RFE (L1-LogReg)')\n",
        "plt.xlabel('RFE Rank (1=Best)')\n",
        "plt.tight_layout()\n",
        "plt.savefig('feature_importance_rfe.png', bbox_inches='tight')\n",
        "plt.close()\n",
        "print('Saved feature_ranks_rfe.csv and feature_importance_rfe.png')"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "retrain_selected"
      },
      "source": [
        "# ================= Retrain on Selected Features (both methods) =================\n",
        "from sklearn.metrics import confusion_matrix\n",
        "import joblib, os\n",
        "\n",
        "def evaluate_on_selected(features, tag):\n",
        "    Xtr = X_train[features]\n",
        "    Xte = X_test[features]\n",
        "    rows = []\n",
        "    os.makedirs(tag, exist_ok=True)\n",
        "    for name, (pipe, grid) in pipelines_and_grids.items():\n",
        "        gs = GridSearchCV(pipe, grid, scoring='roc_auc', cv=cv, n_jobs=-1, refit=True)\n",
        "        gs.fit(Xtr, y_train)\n",
        "        y_proba = gs.best_estimator_.predict_proba(Xte)[:,1]\n",
        "        y_pred  = gs.best_estimator_.predict(Xte)\n",
        "        rows.append({\n",
        "            'model': name,\n",
        "            'features_source': tag,\n",
        "            'num_features': len(features),\n",
        "            'cv_best_auc': round(gs.best_score_,4),\n",
        "            'test_accuracy': round(accuracy_score(y_test, y_pred),4),\n",
        "            'test_precision': round(precision_score(y_test, y_pred),4),\n",
        "            'test_recall': round(recall_score(y_test, y_pred),4),\n",
        "            'test_f1': round(f1_score(y_test, y_pred),4),\n",
        "            'test_roc_auc': round(roc_auc_score(y_test, y_proba),4)\n",
        "        })\n",
        "        # Confusion matrix plot\n",
        "        cm = confusion_matrix(y_test, y_pred)\n",
        "        import matplotlib.pyplot as plt\n",
        "        plt.figure()\n",
        "        plt.imshow(cm, interpolation='nearest')\n",
        "        plt.title(f'Confusion Matrix - {name} ({tag})')\n",
        "        plt.xlabel('Predicted'); plt.ylabel('True')\n",
        "        for (i, j), val in np.ndenumerate(cm):\n",
        "            plt.text(j, i, int(val), ha='center', va='center')\n",
        "        plt.tight_layout()\n",
        "        plt.savefig(f'{tag}/cm_{name}.png', bbox_inches='tight')\n",
        "        plt.close()\n",
        "        joblib.dump(gs.best_estimator_, f'{tag}/best_{name}.joblib')\n",
        "    out = pd.DataFrame(rows).sort_values('test_roc_auc', ascending=False)\n",
        "    out.to_csv(f'{tag}/results_{tag}.csv', index=False)\n",
        "    return out\n",
        "\n",
        "skb_results_df = evaluate_on_selected(selected_features_skb, 'SelectKBest_MI')\n",
        "rfe_results_df = evaluate_on_selected(selected_features_rfe, 'RFE_L1LR')\n",
        "skb_results_df.head(), rfe_results_df.head()"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "final_summary"
      },
      "source": [
        "# ================= Final best model summary =================\n",
        "all_candidates = pd.concat([\n",
        "    pd.read_csv('model_results.csv').assign(features_source='All', num_features=X_train.shape[1]),\n",
        "    pd.read_csv('SelectKBest_MI/results_SelectKBest_MI.csv'),\n",
        "    pd.read_csv('RFE_L1LR/results_RFE_L1LR.csv')\n",
        "], ignore_index=True)\n",
        "\n",
        "best_row = all_candidates.iloc[all_candidates['test_roc_auc'].idxmax()]\n",
        "final_choice = {\n",
        "    'model': best_row['model'],\n",
        "    'features_source': best_row['features_source'],\n",
        "    'num_features': int(best_row['num_features']),\n",
        "    'test_roc_auc': float(best_row['test_roc_auc']),\n",
        "    'test_f1': float(best_row['test_f1']) if 'test_f1' in best_row else None\n",
        "}\n",
        "with open('final_best_model_summary.json','w') as f:\n",
        "    json.dump(final_choice, f, indent=2)\n",
        "final_choice"
      ],
      "outputs": [],
      "execution_count": null
    },
    {
      "cell_type": "code",
      "metadata": {
        "id": "zip_artifacts"
      },
      "source": [
        "# ================= ZIP artifacts for download =================\n",
        "import shutil, os\n",
        "artifacts = [\n",
        "    'EDA_summary.json', 'model_results.csv', 'roc_curves_all_features.png',\n",
        "    'feature_scores_mutual_info.csv', 'feature_importance_mutual_info.png',\n",
        "    'feature_ranks_rfe.csv', 'feature_importance_rfe.png',\n",
        "    'SelectKBest_MI', 'RFE_L1LR', 'final_best_model_summary.json'\n",
        "]\n",
        "zip_name = 'CKD_artifacts_Lama_AlSubaie.zip'\n",
        "with shutil.ZipFile(zip_name, 'w') as z:\n",
        "    pass\n",
        "# Workaround: create via make_archive\n",
        "base_dir = 'ckd_output'\n",
        "os.makedirs(base_dir, exist_ok=True)\n",
        "for a in artifacts:\n",
        "    if os.path.isdir(a):\n",
        "        shutil.copytree(a, os.path.join(base_dir, a), dirs_exist_ok=True)\n",
        "    elif os.path.exists(a):\n",
        "        shutil.copy2(a, os.path.join(base_dir, os.path.basename(a)))\n",
        "shutil.make_archive('CKD_artifacts_Lama_AlSubaie', 'zip', base_dir)\n",
        "print('Created CKD_artifacts_Lama_AlSubaie.zip')"
      ],
      "outputs": [],
      "execution_count": null
    }
  ]
}