{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "865de26a-88b3-4f97-b0df-b8f20d62170d",
   "metadata": {},
   "outputs": [],
   "source": [
    "import os, warnings, numpy as np, pandas as pd\n",
    "warnings.filterwarnings(\"ignore\")\n",
    "\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\n",
    "from sklearn.preprocessing import OneHotEncoder, StandardScaler\n",
    "from sklearn.compose import ColumnTransformer\n",
    "from sklearn.pipeline import Pipeline\n",
    "from sklearn.impute import SimpleImputer\n",
    "from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report, confusion_matrix\n",
    "\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.svm import SVC\n",
    "\n",
    "from sklearn.feature_selection import SelectKBest, mutual_info_classif, SelectFromModel\n",
    "\n",
    "from joblib import dump\n",
    "\n",
    "\n",
    "CSV_PATH = \"Breast_Cancer1.csv\"  \n",
    "TARGET   = \"Status\"             \n",
    "RANDOM_STATE = 42\n",
    "\n",
    "OUT_DIR = \"ml_outputs\"\n",
    "os.makedirs(OUT_DIR, exist_ok=True)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "456e061a-19cc-4a9d-b5a1-d67d7890dccb",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "Age               0\n",
       "Race              0\n",
       "Marital Status    0\n",
       "T Stage           0\n",
       "N Stage           0\n",
       "6th Stage         0\n",
       "differentiate     0\n",
       "Grade             0\n",
       "A Stage           0\n",
       "Tumor Size        0\n",
       "dtype: int64"
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    }
   ],
   "source": [
    "df = pd.read_csv(CSV_PATH)\n",
    "assert TARGET in df.columns, f\"Target '{TARGET}' not found! Available: {list(df.columns)}\"\n",
    "\n",
    "y = df[TARGET]\n",
    "X = df.drop(columns=[TARGET])\n",
    "\n",
    "num_cols = X.select_dtypes(include=[np.number]).columns.tolist()\n",
    "cat_cols = [c for c in X.columns if c not in num_cols]\n",
    "\n",
    "num_imputer = SimpleImputer(strategy=\"median\")\n",
    "cat_imputer = SimpleImputer(strategy=\"most_frequent\")\n",
    "\n",
    "if num_cols:\n",
    "    X[num_cols] = num_imputer.fit_transform(X[num_cols])\n",
    "if cat_cols:\n",
    "    X[cat_cols] = cat_imputer.fit_transform(X[cat_cols])\n",
    "\n",
    "df_clean = pd.concat([X, y], axis=1)\n",
    "df_clean.to_csv(CSV_PATH, index=False)\n",
    "\n",
    "\n",
    "display(df_clean.isnull().sum().sort_values(ascending=False).head(10))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "50ab35c7-1359-4f44-8daa-6a904a3e405a",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>phase</th>\n",
       "      <th>model</th>\n",
       "      <th>best_params</th>\n",
       "      <th>cv_best_f1_macro</th>\n",
       "      <th>accuracy</th>\n",
       "      <th>precision_macro</th>\n",
       "      <th>recall_macro</th>\n",
       "      <th>f1_macro</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>full</td>\n",
       "      <td>LogReg</td>\n",
       "      <td>{'clf__C': 1.0, 'clf__penalty': 'l1'}</td>\n",
       "      <td>0.758953</td>\n",
       "      <td>0.895652</td>\n",
       "      <td>0.834795</td>\n",
       "      <td>0.718511</td>\n",
       "      <td>0.758134</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>full</td>\n",
       "      <td>RandomForest</td>\n",
       "      <td>{'clf__max_depth': None, 'clf__min_samples_spl...</td>\n",
       "      <td>0.790983</td>\n",
       "      <td>0.894410</td>\n",
       "      <td>0.838776</td>\n",
       "      <td>0.707782</td>\n",
       "      <td>0.749800</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>full</td>\n",
       "      <td>SVC</td>\n",
       "      <td>{'clf__C': 10.0, 'clf__gamma': 'scale', 'clf__...</td>\n",
       "      <td>0.741148</td>\n",
       "      <td>0.881988</td>\n",
       "      <td>0.793408</td>\n",
       "      <td>0.693787</td>\n",
       "      <td>0.727642</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "  phase         model                                        best_params  \\\n",
       "0  full        LogReg              {'clf__C': 1.0, 'clf__penalty': 'l1'}   \n",
       "1  full  RandomForest  {'clf__max_depth': None, 'clf__min_samples_spl...   \n",
       "2  full           SVC  {'clf__C': 10.0, 'clf__gamma': 'scale', 'clf__...   \n",
       "\n",
       "   cv_best_f1_macro  accuracy  precision_macro  recall_macro  f1_macro  \n",
       "0          0.758953  0.895652         0.834795      0.718511  0.758134  \n",
       "1          0.790983  0.894410         0.838776      0.707782  0.749800  \n",
       "2          0.741148  0.881988         0.793408      0.693787  0.727642  "
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    }
   ],
   "source": [
    "df = pd.read_csv(CSV_PATH)\n",
    "y = df[TARGET]\n",
    "X = df.drop(columns=[TARGET])\n",
    "\n",
    "num_cols = X.select_dtypes(include=[np.number]).columns.tolist()\n",
    "cat_cols = [c for c in X.columns if c not in num_cols]\n",
    "\n",
    "numeric_tf = Pipeline([(\"imputer\", SimpleImputer(strategy=\"median\")), (\"scaler\", StandardScaler())])\n",
    "categorical_tf = Pipeline([(\"imputer\", SimpleImputer(strategy=\"most_frequent\")), (\"onehot\", OneHotEncoder(handle_unknown=\"ignore\"))])\n",
    "\n",
    "preprocessor = ColumnTransformer([(\"num\", numeric_tf, num_cols), (\"cat\", categorical_tf, cat_cols)])\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, stratify=y, random_state=RANDOM_STATE)\n",
    "\n",
    "models = {\n",
    "    \"LogReg\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", LogisticRegression(max_iter=2000, solver=\"liblinear\", random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__penalty\": [\"l1\",\"l2\"], \"clf__C\": [0.1, 1.0, 10.0]}\n",
    "    },\n",
    "    \"RandomForest\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", RandomForestClassifier(random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__n_estimators\": [200, 400], \"clf__max_depth\": [None, 10], \"clf__min_samples_split\": [2, 5]}\n",
    "    },\n",
    "    \"SVC\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", SVC(probability=False, random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__kernel\": [\"rbf\",\"linear\"], \"clf__C\": [1.0, 10.0], \"clf__gamma\": [\"scale\",\"auto\"]}\n",
    "    }\n",
    "}\n",
    "\n",
    "cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=RANDOM_STATE)\n",
    "\n",
    "def fit_grid_eval(name, pipe, grid):\n",
    "    gs = GridSearchCV(pipe, grid, scoring=\"f1_macro\", cv=cv, n_jobs=-1, verbose=0)\n",
    "    gs.fit(X_train, y_train)\n",
    "    best = gs.best_estimator_\n",
    "    y_pred = best.predict(X_test)\n",
    "    row = {\n",
    "        \"phase\": \"full\",\n",
    "        \"model\": name,\n",
    "        \"best_params\": str(gs.best_params_),\n",
    "        \"cv_best_f1_macro\": gs.best_score_,\n",
    "        \"accuracy\": accuracy_score(y_test, y_pred),\n",
    "        \"precision_macro\": precision_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "        \"recall_macro\": recall_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "        \"f1_macro\": f1_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "    }\n",
    "    dump(best, os.path.join(OUT_DIR, f\"best_full_{name}.joblib\"))\n",
    "    with open(os.path.join(OUT_DIR, f\"report_full_{name}.txt\"), \"w\", encoding=\"utf-8\") as f:\n",
    "        f.write(classification_report(y_test, y_pred, zero_division=0))\n",
    "    return row, confusion_matrix(y_test, y_pred)\n",
    "\n",
    "full_rows, cms_full = [], {}\n",
    "for name, spec in models.items():\n",
    "    r, cm = fit_grid_eval(name, spec[\"pipe\"], spec[\"grid\"])\n",
    "    full_rows.append(r); cms_full[name] = cm\n",
    "\n",
    "df_full = pd.DataFrame(full_rows).sort_values(\"f1_macro\", ascending=False)\n",
    "display(df_full)\n",
    "df_full.to_csv(os.path.join(OUT_DIR, \"results_full.csv\"), index=False)\n",
    "\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "1b22123f-3903-4956-8fbb-a9bf9a1d5a87",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "KBest: ['Survival Months', 'N Stage', 'Reginol Node Positive']\n",
      "L1: ['Survival Months', 'N Stage']\n",
      "Final selected (intersection): ['N Stage', 'Survival Months']\n"
     ]
    }
   ],
   "source": [
    "\n",
    "df = pd.read_csv(CSV_PATH)\n",
    "y = df[TARGET]\n",
    "X = df.drop(columns=[TARGET])\n",
    "\n",
    "num_cols = X.select_dtypes(include=[np.number]).columns.tolist()\n",
    "cat_cols = [c for c in X.columns if c not in num_cols]\n",
    "\n",
    "num_imputer = SimpleImputer(strategy=\"median\")\n",
    "cat_imputer = SimpleImputer(strategy=\"most_frequent\")\n",
    "if num_cols:\n",
    "    X[num_cols] = num_imputer.fit_transform(X[num_cols])\n",
    "if cat_cols:\n",
    "    X[cat_cols] = cat_imputer.fit_transform(X[cat_cols])\n",
    "\n",
    "X_cat_oh = pd.get_dummies(X[cat_cols], drop_first=False) if cat_cols else pd.DataFrame(index=X.index)\n",
    "X_all = pd.concat([X[num_cols], X_cat_oh], axis=1)\n",
    "\n",
    "orig_to_expanded = {}\n",
    "for c in num_cols:\n",
    "    orig_to_expanded[c] = [c]\n",
    "for c in cat_cols:\n",
    "    cols = [col for col in X_all.columns if col.startswith(c + \"_\")]\n",
    "    if not cols:\n",
    "        cols = [c]\n",
    "    orig_to_expanded[c] = cols\n",
    "\n",
    "# (1) SelectKBest(mutual_info)\n",
    "y_enc = y.astype(\"category\").cat.codes if not np.issubdtype(y.dtype, np.number) else y\n",
    "# k يعتمد على عدد الأعمدة الأصلية\n",
    "TOP_K = min(15, max(5, len(X.columns)//3))\n",
    "skb = SelectKBest(mutual_info_classif, k=min(TOP_K, X_all.shape[1]))\n",
    "skb.fit(X_all, y_enc)\n",
    "scores = pd.Series(skb.scores_, index=X_all.columns).fillna(0)\n",
    "kbest_scores_by_orig = {orig: float(scores.loc[[c for c in orig_to_expanded[orig] if c in scores]].sum()) for orig in X.columns}\n",
    "kbest_selected = list(pd.Series(kbest_scores_by_orig).sort_values(ascending=False).head(TOP_K).index)\n",
    "\n",
    "# (2) L1-Logistic (SelectFromModel) \n",
    "scaler = StandardScaler(with_mean=False)\n",
    "X_all_scaled = pd.DataFrame(scaler.fit_transform(X_all), columns=X_all.columns, index=X_all.index)\n",
    "\n",
    "l1_lr = LogisticRegression(penalty=\"l1\", solver=\"liblinear\", max_iter=3000, random_state=RANDOM_STATE)\n",
    "l1_lr.fit(X_all_scaled, y_enc)\n",
    "coefs = l1_lr.coef_\n",
    "abs_coef = np.abs(coefs).mean(axis=0) if coefs.ndim == 2 else np.abs(coefs)\n",
    "abs_coef_s = pd.Series(abs_coef.ravel(), index=X_all.columns)\n",
    "\n",
    "l1_scores_by_orig = {orig: float(abs_coef_s.loc[[c for c in orig_to_expanded[orig] if c in abs_coef_s]].sum()) for orig in X.columns}\n",
    "l1_rank = pd.Series(l1_scores_by_orig).sort_values(ascending=False)\n",
    "median_thr = l1_rank.median()\n",
    "l1_selected = list(l1_rank[l1_rank >= median_thr].index)\n",
    "if len(l1_selected) > TOP_K:\n",
    "    l1_selected = list(l1_rank.head(TOP_K).index)\n",
    "\n",
    "final_selected = sorted(list(set(kbest_selected).intersection(l1_selected)))\n",
    "if len(final_selected) == 0:  # احتياط\n",
    "    final_selected = kbest_selected[:]\n",
    "\n",
    "print(\"KBest:\", kbest_selected)\n",
    "print(\"L1:\", l1_selected)\n",
    "print(\"Final selected (intersection):\", final_selected)\n",
    "\n",
    "df_selected = pd.concat([X[final_selected], y], axis=1)\n",
    "df_selected.to_csv(CSV_PATH, index=False)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "id": "e4d1ae48-f075-4e00-bf8a-6c774e38057e",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>phase</th>\n",
       "      <th>model</th>\n",
       "      <th>best_params</th>\n",
       "      <th>cv_best_f1_macro</th>\n",
       "      <th>accuracy</th>\n",
       "      <th>precision_macro</th>\n",
       "      <th>recall_macro</th>\n",
       "      <th>f1_macro</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>selected</td>\n",
       "      <td>RandomForest</td>\n",
       "      <td>{'clf__max_depth': 10, 'clf__min_samples_split...</td>\n",
       "      <td>0.752813</td>\n",
       "      <td>0.888199</td>\n",
       "      <td>0.800956</td>\n",
       "      <td>0.724108</td>\n",
       "      <td>0.753571</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>selected</td>\n",
       "      <td>SVC</td>\n",
       "      <td>{'clf__C': 10.0, 'clf__gamma': 'scale', 'clf__...</td>\n",
       "      <td>0.767011</td>\n",
       "      <td>0.883230</td>\n",
       "      <td>0.798037</td>\n",
       "      <td>0.694520</td>\n",
       "      <td>0.729340</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>selected</td>\n",
       "      <td>LogReg</td>\n",
       "      <td>{'clf__C': 10.0, 'clf__penalty': 'l1'}</td>\n",
       "      <td>0.738721</td>\n",
       "      <td>0.879503</td>\n",
       "      <td>0.796387</td>\n",
       "      <td>0.672329</td>\n",
       "      <td>0.709255</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "      phase         model                                        best_params  \\\n",
       "1  selected  RandomForest  {'clf__max_depth': 10, 'clf__min_samples_split...   \n",
       "2  selected           SVC  {'clf__C': 10.0, 'clf__gamma': 'scale', 'clf__...   \n",
       "0  selected        LogReg             {'clf__C': 10.0, 'clf__penalty': 'l1'}   \n",
       "\n",
       "   cv_best_f1_macro  accuracy  precision_macro  recall_macro  f1_macro  \n",
       "1          0.752813  0.888199         0.800956      0.724108  0.753571  \n",
       "2          0.767011  0.883230         0.798037      0.694520  0.729340  \n",
       "0          0.738721  0.879503         0.796387      0.672329  0.709255  "
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    }
   ],
   "source": [
    "df = pd.read_csv(CSV_PATH)\n",
    "y = df[TARGET]\n",
    "X = df.drop(columns=[TARGET])\n",
    "\n",
    "#  preprocessor\n",
    "num_cols = X.select_dtypes(include=[np.number]).columns.tolist()\n",
    "cat_cols = [c for c in X.columns if c not in num_cols]\n",
    "\n",
    "numeric_tf = Pipeline([(\"imputer\", SimpleImputer(strategy=\"median\")), (\"scaler\", StandardScaler())])\n",
    "categorical_tf = Pipeline([(\"imputer\", SimpleImputer(strategy=\"most_frequent\")), (\"onehot\", OneHotEncoder(handle_unknown=\"ignore\"))])\n",
    "\n",
    "preprocessor = ColumnTransformer([(\"num\", numeric_tf, num_cols), (\"cat\", categorical_tf, cat_cols)])\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, stratify=y, random_state=RANDOM_STATE)\n",
    "\n",
    "models = {\n",
    "    \"LogReg\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", LogisticRegression(max_iter=2000, solver=\"liblinear\", random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__penalty\": [\"l1\",\"l2\"], \"clf__C\": [0.1, 1.0, 10.0]}\n",
    "    },\n",
    "    \"RandomForest\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", RandomForestClassifier(random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__n_estimators\": [200, 400], \"clf__max_depth\": [None, 10], \"clf__min_samples_split\": [2, 5]}\n",
    "    },\n",
    "    \"SVC\": {\n",
    "        \"pipe\": Pipeline([(\"prep\", preprocessor), (\"clf\", SVC(probability=False, random_state=RANDOM_STATE))]),\n",
    "        \"grid\": {\"clf__kernel\": [\"rbf\",\"linear\"], \"clf__C\": [1.0, 10.0], \"clf__gamma\": [\"scale\",\"auto\"]}\n",
    "    }\n",
    "}\n",
    "cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=RANDOM_STATE)\n",
    "\n",
    "def fit_grid_eval_selected(name, pipe, grid):\n",
    "    gs = GridSearchCV(pipe, grid, scoring=\"f1_macro\", cv=cv, n_jobs=-1, verbose=0)\n",
    "    gs.fit(X_train, y_train)\n",
    "    best = gs.best_estimator_\n",
    "    y_pred = best.predict(X_test)\n",
    "    row = {\n",
    "        \"phase\": \"selected\",\n",
    "        \"model\": name,\n",
    "        \"best_params\": str(gs.best_params_),\n",
    "        \"cv_best_f1_macro\": gs.best_score_,\n",
    "        \"accuracy\": accuracy_score(y_test, y_pred),\n",
    "        \"precision_macro\": precision_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "        \"recall_macro\": recall_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "        \"f1_macro\": f1_score(y_test, y_pred, average=\"macro\", zero_division=0),\n",
    "    }\n",
    "    dump(best, os.path.join(OUT_DIR, f\"best_selected_{name}.joblib\"))\n",
    "    with open(os.path.join(OUT_DIR, f\"report_selected_{name}.txt\"), \"w\", encoding=\"utf-8\") as f:\n",
    "        f.write(classification_report(y_test, y_pred, zero_division=0))\n",
    "    return row, confusion_matrix(y_test, y_pred)\n",
    "\n",
    "sel_rows, cms_sel = [], {}\n",
    "for name, spec in models.items():\n",
    "    r, cm = fit_grid_eval_selected(name, spec[\"pipe\"], spec[\"grid\"])\n",
    "    sel_rows.append(r); cms_sel[name] = cm\n",
    "\n",
    "df_selected_phase = pd.DataFrame(sel_rows).sort_values(\"f1_macro\", ascending=False)\n",
    "display(df_selected_phase)\n",
    "df_selected_phase.to_csv(os.path.join(OUT_DIR, \"results_selected.csv\"), index=False)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "6fc7937c-b7d5-4c50-8dc0-1ae7bf03694d",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3 (ipykernel)",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.11"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
