{
 "cells": [
  {
   "cell_type": "markdown",
   "id": "ce9f6954-76e7-421d-85d1-3b80aa4ef47b",
   "metadata": {},
   "source": [
    "<div style=\"text-align: center; font-family: Arial, sans-serif;\">\n",
    "\n",
    "  <h1 style=\"color: navy; font-size: 36px; font-weight: bold;\">\n",
    "    Practical Image Processing and Natural Language Processing\n",
    "  </h1>\n",
    "\n",
    "  <h3 style=\"color:black darkred; font-size: 28px;\">\n",
    "    Waleed Mohammed Rasheedy\n",
    "  </h3>\n",
    "\n",
    "  <h3 style=\"color: black; font-size: 24px;\">\n",
    "    2021204005\n",
    "  </h3>\n",
    "\n",
    "  <h2 style=\"color:black darkred; font-size: 28px;\">\n",
    "    Master’s Student, Master of Science in Artificial Intelligence Updated\n",
    "   </h2>\n",
    "\n",
    "   <h2 style=\"color:black darkred; font-size: 28px;\">\n",
    "    College of Informatics, Midocean University\n",
    "   </h2>\n",
    "\n",
    "   <h2 style=\"color:black darkred; font-size: 28px;\">\n",
    "    Under Supervision of<strong> Dr. Hager Saleh</strong>\n",
    "   </h2>\n",
    "\n",
    "</div>"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "08926327-438f-446c-a75d-a8b46da4f8b2",
   "metadata": {},
   "source": [
    "## Import necessary libraries"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "4bad4751-c49d-4023-b44a-a86d44dd6997",
   "metadata": {},
   "outputs": [],
   "source": [
    " import pandas as pd\n",
    " import numpy as np\n",
    " from sklearn.model_selection import train_test_split, GridSearchCV\n",
    " from sklearn.preprocessing import StandardScaler, LabelEncoder\n",
    " from sklearn.linear_model import LogisticRegression\n",
    " from sklearn.tree import DecisionTreeClassifier\n",
    " from sklearn.ensemble import RandomForestClassifier\n",
    " from sklearn.svm import SVC\n",
    " from sklearn.naive_bayes import GaussianNB\n",
    " from sklearn.neighbors import KNeighborsClassifier\n",
    " from sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, classification_report\n",
    " import matplotlib.pyplot as plt\n",
    " import seaborn as sns\n",
    " from sklearn.model_selection import GridSearchCV\n",
    " from sklearn.linear_model import LogisticRegression\n",
    " from sklearn.metrics import classification_report\n",
    " from sklearn.feature_extraction.text import TfidfVectorizer\n",
    " from sklearn.model_selection import train_test_split\n",
    " from sklearn.feature_extraction.text import CountVectorizer\n",
    " import re\n",
    " import nltk\n",
    " import spacy\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "ad4192b8-0e2f-4f1e-9102-c66a2af34428",
   "metadata": {},
   "outputs": [
    {
     "name": "stderr",
     "output_type": "stream",
     "text": [
      "[nltk_data] Downloading package punkt to\n",
      "[nltk_data]     C:\\Users\\Waleed\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package punkt is already up-to-date!\n",
      "[nltk_data] Downloading package stopwords to\n",
      "[nltk_data]     C:\\Users\\Waleed\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package stopwords is already up-to-date!\n",
      "[nltk_data] Downloading package wordnet to\n",
      "[nltk_data]     C:\\Users\\Waleed\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package wordnet is already up-to-date!\n"
     ]
    },
    {
     "data": {
      "text/plain": [
       "True"
      ]
     },
     "execution_count": 2,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    " from nltk.corpus import stopwords\n",
    " from nltk.tokenize import word_tokenize\n",
    " from nltk.stem import PorterStemmer\n",
    " from nltk.stem import WordNetLemmatizer\n",
    " # Download NLTK resources (only first time)\n",
    " nltk.download('punkt')\n",
    " nltk.download('stopwords')\n",
    " nltk.download('wordnet')"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "d76beb18-180e-49b5-9896-a9af1b243c34",
   "metadata": {},
   "source": [
    "## Function to evaluate model"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "d06e352d-713f-4b97-9ff5-6d6b28825350",
   "metadata": {},
   "outputs": [],
   "source": [
    "def evaluate_model(model_name, y_true, y_pred):\n",
    "    \n",
    "    # Calculate metrics\n",
    "    accuracy = accuracy_score(y_true, y_pred)\n",
    "    precision = precision_score(y_true, y_pred, average='weighted')  \n",
    "    recall = recall_score(y_true, y_pred, average='weighted')\n",
    "    f1 = f1_score(y_true, y_pred, average='weighted')\n",
    "    cm = confusion_matrix(y_true, y_pred)\n",
    "    \n",
    "    # Create a report\n",
    "    report = classification_report(y_true, y_pred)\n",
    "        \n",
    "    # Output results\n",
    "    metrics = {\n",
    "        'Model Name': model_name,\n",
    "        'Accuracy': accuracy,\n",
    "        'Precision': precision,\n",
    "        'Recall': recall,\n",
    "        'F1 Score': f1,\n",
    "       \n",
    "        'Classification Report': report\n",
    "    }\n",
    " \n",
    "    return metrics"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "b662b0c3-7ee2-4ed4-9831-c90c5e3d18bf",
   "metadata": {},
   "source": [
    "## Function to apply Random forest with grid search"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "id": "363074ec-028e-45de-b312-bf43c6788410",
   "metadata": {},
   "outputs": [],
   "source": [
    " from sklearn.ensemble import RandomForestClassifier\n",
    " from sklearn.model_selection import GridSearchCV\n",
    " def train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model_func):\n",
    "    # Define the Random Forest Classifier\n",
    "    rf_classifier = RandomForestClassifier(random_state=42)\n",
    "    # Define the hyperparameters for grid search\n",
    "    param_grid = {\n",
    "        'n_estimators': [100, 200, 300],\n",
    "        'max_depth': [None, 10, 20, 30],\n",
    "    }\n",
    "    # Initialize GridSearchCV\n",
    "    grid_search = GridSearchCV(estimator=rf_classifier, param_grid=param_grid, cv=5,\n",
    "                               scoring='accuracy', n_jobs=-1)\n",
    "    # Fit the grid search to the data\n",
    "    grid_search.fit(X_train_vec, y_train)\n",
    "    # Get the best estimator from grid search\n",
    "    best_rf_classifier = grid_search.best_estimator_\n",
    "    # Make predictions using the best model\n",
    "    y_pred = best_rf_classifier.predict(X_test_vec)\n",
    "    # Evaluate the model\n",
    "    evaluation_results = evaluate_model_func('RandomForestClassifier', y_test, y_pred)\n",
    "    # Print the evaluation results\n",
    "    for key, value in evaluation_results.items():\n",
    "        if key == 'Classification Report':\n",
    "            print(value)  # Print report separately for better readability\n",
    "        else:\n",
    "            print(f\"{key}: {value:.4f}\" if isinstance(value, float) \n",
    "        else f\"{key}: \\n{value}\")\n",
    "            # Print the best parameters found by grid search\n",
    "            print(\"\\nBest hyperparameters found by GridSearchCV:\")\n",
    "            print(grid_search.best_params_)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "5a2c4e15-a68d-455b-a8eb-cd9161e959d1",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>author</th>\n",
       "      <th>published</th>\n",
       "      <th>title</th>\n",
       "      <th>text</th>\n",
       "      <th>language</th>\n",
       "      <th>site_url</th>\n",
       "      <th>main_img_url</th>\n",
       "      <th>type</th>\n",
       "      <th>label</th>\n",
       "      <th>title_without_stopwords</th>\n",
       "      <th>text_without_stopwords</th>\n",
       "      <th>hasImage</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>Barracuda Brigade</td>\n",
       "      <td>2016-10-26T21:41:00.000+03:00</td>\n",
       "      <td>muslims busted they stole millions in govt ben...</td>\n",
       "      <td>print they should pay all the back all the mon...</td>\n",
       "      <td>english</td>\n",
       "      <td>100percentfedup.com</td>\n",
       "      <td>http://bb4sp.com/wp-content/uploads/2016/10/Fu...</td>\n",
       "      <td>bias</td>\n",
       "      <td>Real</td>\n",
       "      <td>muslims busted stole millions govt benefits</td>\n",
       "      <td>print pay back money plus interest entire fami...</td>\n",
       "      <td>1.0</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>reasoning with facts</td>\n",
       "      <td>2016-10-29T08:47:11.259+03:00</td>\n",
       "      <td>re why did attorney general loretta lynch plea...</td>\n",
       "      <td>why did attorney general loretta lynch plead t...</td>\n",
       "      <td>english</td>\n",
       "      <td>100percentfedup.com</td>\n",
       "      <td>http://bb4sp.com/wp-content/uploads/2016/10/Fu...</td>\n",
       "      <td>bias</td>\n",
       "      <td>Real</td>\n",
       "      <td>attorney general loretta lynch plead fifth</td>\n",
       "      <td>attorney general loretta lynch plead fifth bar...</td>\n",
       "      <td>1.0</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>Barracuda Brigade</td>\n",
       "      <td>2016-10-31T01:41:49.479+02:00</td>\n",
       "      <td>breaking weiner cooperating with fbi on hillar...</td>\n",
       "      <td>red state  \\nfox news sunday reported this mor...</td>\n",
       "      <td>english</td>\n",
       "      <td>100percentfedup.com</td>\n",
       "      <td>http://bb4sp.com/wp-content/uploads/2016/10/Fu...</td>\n",
       "      <td>bias</td>\n",
       "      <td>Real</td>\n",
       "      <td>breaking weiner cooperating fbi hillary email ...</td>\n",
       "      <td>red state fox news sunday reported morning ant...</td>\n",
       "      <td>1.0</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>3</th>\n",
       "      <td>Fed Up</td>\n",
       "      <td>2016-11-01T05:22:00.000+02:00</td>\n",
       "      <td>pin drop speech by father of daughter kidnappe...</td>\n",
       "      <td>email kayla mueller was a prisoner and torture...</td>\n",
       "      <td>english</td>\n",
       "      <td>100percentfedup.com</td>\n",
       "      <td>http://100percentfedup.com/wp-content/uploads/...</td>\n",
       "      <td>bias</td>\n",
       "      <td>Real</td>\n",
       "      <td>pin drop speech father daughter kidnapped kill...</td>\n",
       "      <td>email kayla mueller prisoner tortured isis cha...</td>\n",
       "      <td>1.0</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>4</th>\n",
       "      <td>Fed Up</td>\n",
       "      <td>2016-11-01T21:56:00.000+02:00</td>\n",
       "      <td>fantastic trumps  point plan to reform healthc...</td>\n",
       "      <td>email healthcare reform to make america great ...</td>\n",
       "      <td>english</td>\n",
       "      <td>100percentfedup.com</td>\n",
       "      <td>http://100percentfedup.com/wp-content/uploads/...</td>\n",
       "      <td>bias</td>\n",
       "      <td>Real</td>\n",
       "      <td>fantastic trumps point plan reform healthcare ...</td>\n",
       "      <td>email healthcare reform make america great sin...</td>\n",
       "      <td>1.0</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "                 author                      published  \\\n",
       "0     Barracuda Brigade  2016-10-26T21:41:00.000+03:00   \n",
       "1  reasoning with facts  2016-10-29T08:47:11.259+03:00   \n",
       "2     Barracuda Brigade  2016-10-31T01:41:49.479+02:00   \n",
       "3                Fed Up  2016-11-01T05:22:00.000+02:00   \n",
       "4                Fed Up  2016-11-01T21:56:00.000+02:00   \n",
       "\n",
       "                                               title  \\\n",
       "0  muslims busted they stole millions in govt ben...   \n",
       "1  re why did attorney general loretta lynch plea...   \n",
       "2  breaking weiner cooperating with fbi on hillar...   \n",
       "3  pin drop speech by father of daughter kidnappe...   \n",
       "4  fantastic trumps  point plan to reform healthc...   \n",
       "\n",
       "                                                text language  \\\n",
       "0  print they should pay all the back all the mon...  english   \n",
       "1  why did attorney general loretta lynch plead t...  english   \n",
       "2  red state  \\nfox news sunday reported this mor...  english   \n",
       "3  email kayla mueller was a prisoner and torture...  english   \n",
       "4  email healthcare reform to make america great ...  english   \n",
       "\n",
       "              site_url                                       main_img_url  \\\n",
       "0  100percentfedup.com  http://bb4sp.com/wp-content/uploads/2016/10/Fu...   \n",
       "1  100percentfedup.com  http://bb4sp.com/wp-content/uploads/2016/10/Fu...   \n",
       "2  100percentfedup.com  http://bb4sp.com/wp-content/uploads/2016/10/Fu...   \n",
       "3  100percentfedup.com  http://100percentfedup.com/wp-content/uploads/...   \n",
       "4  100percentfedup.com  http://100percentfedup.com/wp-content/uploads/...   \n",
       "\n",
       "   type label                            title_without_stopwords  \\\n",
       "0  bias  Real        muslims busted stole millions govt benefits   \n",
       "1  bias  Real         attorney general loretta lynch plead fifth   \n",
       "2  bias  Real  breaking weiner cooperating fbi hillary email ...   \n",
       "3  bias  Real  pin drop speech father daughter kidnapped kill...   \n",
       "4  bias  Real  fantastic trumps point plan reform healthcare ...   \n",
       "\n",
       "                              text_without_stopwords  hasImage  \n",
       "0  print pay back money plus interest entire fami...       1.0  \n",
       "1  attorney general loretta lynch plead fifth bar...       1.0  \n",
       "2  red state fox news sunday reported morning ant...       1.0  \n",
       "3  email kayla mueller prisoner tortured isis cha...       1.0  \n",
       "4  email healthcare reform make america great sin...       1.0  "
      ]
     },
     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    " df = pd.read_csv(\"news_articles.csv\") \n",
    " df.head()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "45dee912-e039-4adf-af3b-9abb87b15039",
   "metadata": {},
   "source": [
    "## Function to Automatically Download Missing NLTK Data Packages"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "id": "45445443-5722-4794-804e-cc0ac6a721d2",
   "metadata": {},
   "outputs": [],
   "source": [
    "import nltk\n",
    "\n",
    "def ensure_nltk_resources(pkgs):\n",
    "    for name, path in pkgs:\n",
    "        try:\n",
    "            nltk.data.find(path)\n",
    "        except LookupError:\n",
    "            nltk.download(name, quiet=True)\n",
    "\n",
    "ensure_nltk_resources([\n",
    "    (\"punkt\", \"tokenizers/punkt\"),\n",
    "    (\"punkt_tab\", \"tokenizers/punkt_tab\"), \n",
    "    (\"stopwords\", \"corpora/stopwords\"),    \n",
    "    (\"wordnet\", \"corpora/wordnet\"),        \n",
    "    (\"omw-1.4\", \"corpora/omw-1.4\"),        \n",
    "])"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "c5988651-41a8-49f5-b02d-9677036f01c9",
   "metadata": {},
   "source": [
    "## Preprocessing English text"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "00e4e116-c301-4b0e-a12e-6fc0f77c4fdc",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "0       print pay back money plu interest entir famili...\n",
       "1       attorney gener loretta lynch plead fifth barra...\n",
       "2       red state fox news sunday report morn anthoni ...\n",
       "3       email kayla mueller prison tortur isi chanc re...\n",
       "4       email healthcar reform make america great sinc...\n",
       "                              ...                        \n",
       "2041    prof cano reek genocid white privileg craft lo...\n",
       "2042    teen walk free gangrap convict judg said group...\n",
       "2043    school name munichmassacr mastermind terrorist...\n",
       "2044    war rumor war russia unveil satan missil nucle...\n",
       "2045    check hillarythem haunt hous anticlinton yard ...\n",
       "Name: clean_text, Length: 2045, dtype: object"
      ]
     },
     "execution_count": 7,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    " stemmer = PorterStemmer()\n",
    " stop_words = set(stopwords.words('english'))\n",
    " def preprocess_text(text):\n",
    "    text = text.lower()\n",
    "    # 2. Remove URLs\n",
    "    text = re.sub(r'http\\S+|www.\\S+', '', text)\n",
    "    # 3. Remove HTML tags\n",
    "    text = re.sub(r'<.*?>', '', text)\n",
    " \n",
    "    # 4. Remove Special Characters, Numbers, Punctuation\n",
    "    text = re.sub(r'[^a-zA-Z\\s]', '', text)\n",
    "    # 5. Tokenization\n",
    "    tokens = word_tokenize(text)\n",
    "    # 6. Remove Stop Words\n",
    "    tokens = [word for word in tokens if word not in stop_words]\n",
    "      # 7. Stemming\n",
    "    stemmed_tokens = [stemmer.stem(word) for word in tokens]\n",
    "       # 8. Final clean text reconstruction (optional)\n",
    "    clean_text = ' '.join(stemmed_tokens)\n",
    "  \n",
    "    return clean_text\n",
    "\n",
    " df = df.dropna(subset=['text'])\n",
    " df[\"clean_text\"] = df[\"text\"].apply(preprocess_text)\n",
    "\n",
    " df[\"clean_text\"] = df[\"text\"].apply(preprocess_text)\n",
    " df = df.dropna()\n",
    " df['clean_text']\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "id": "4465b2ef-a63f-42e8-afbb-83a9a1a24709",
   "metadata": {},
   "outputs": [],
   "source": [
    " df.head()\n",
    " X = df['clean_text']\n",
    " y = df['label']"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "e551093d-2f1b-42ea-b2e5-a379a628f3eb",
   "metadata": {},
   "source": [
    "## Studying the Effect of Applying ML Models on Imbalanced Datasets"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "id": "2337f88b-9c6c-43cb-81c9-0c0b3acc3f98",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train, X_test, y_train, y_test = train_test_split( X, y, test_size=0.30, random_state=42, stratify=y)\n",
    "vectorizer = TfidfVectorizer()\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec = vectorizer.transform(X_test)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "8fb0f171-44e7-462c-9e8a-85ea995d348b",
   "metadata": {},
   "source": [
    "## Random Forest Classifier"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 10,
   "id": "903fdeeb-b9f1-4f3c-b691-4c565f48887f",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Accuracy: 0.7541\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Precision: 0.7848\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Recall: 0.7541\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "F1 Score: 0.7233\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "        Fake       0.73      0.97      0.83       388\n",
      "        Real       0.88      0.38      0.54       226\n",
      "\n",
      "    accuracy                           0.75       614\n",
      "   macro avg       0.80      0.68      0.68       614\n",
      "weighted avg       0.78      0.75      0.72       614\n",
      "\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "e7d4b531-825f-4d35-904b-17c1809cb20f",
   "metadata": {},
   "source": [
    "## Example of each different method to handle an imbalanced dataset"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 11,
   "id": "f897f59a-d9b7-4478-ac53-bb40a95f1d59",
   "metadata": {},
   "outputs": [],
   "source": [
    " X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=42, stratify=y)\n",
    " vectorizer = TfidfVectorizer()\n",
    " X_train_vec = vectorizer.fit_transform(X_train)\n",
    " X_test_vec = vectorizer.transform(X_test)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "b04ea60d-6eb5-4046-92e5-045a5399f520",
   "metadata": {},
   "source": [
    "## RandomOverSampler"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 12,
   "id": "651ad332-1b74-4886-8a30-ae1c8e4905ba",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After RandomOversampling: Counter({'Fake': 903, 'Real': 903})\n"
     ]
    }
   ],
   "source": [
    " from imblearn.over_sampling import RandomOverSampler\n",
    " from collections import Counter\n",
    " ros = RandomOverSampler(random_state=42)\n",
    " X_ros, y_ros = ros.fit_resample(X_train_vec, y_train)\n",
    " print(\"After RandomOversampling:\", Counter(y_ros))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 28,
   "id": "4fe5d686-4140-4d39-9352-b6bdf6dece64",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Accuracy: 0.7541\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Precision: 0.7848\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Recall: 0.7541\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "F1 Score: 0.7233\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "        Fake       0.73      0.97      0.83       388\n",
      "        Real       0.88      0.38      0.54       226\n",
      "\n",
      "    accuracy                           0.75       614\n",
      "   macro avg       0.80      0.68      0.68       614\n",
      "weighted avg       0.78      0.75      0.72       614\n",
      "\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model)\n"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "0031e57c-842b-4fa9-a4a9-b1006925ec5c",
   "metadata": {},
   "source": [
    "## RandomUnderSampler"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "id": "8226713e-182c-43d9-836d-424cd38a0fa7",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After undersampling: Counter({'Fake': 528, 'Real': 528})\n"
     ]
    }
   ],
   "source": [
    "from imblearn.under_sampling import RandomUnderSampler\n",
    "rus = RandomUnderSampler(random_state=42)\n",
    "X_res, y_res = rus.fit_resample(X_train_vec, y_train)\n",
    "print(\"After undersampling:\", Counter(y_res))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 14,
   "id": "30805270-8c7d-42d2-aad8-66aa84475304",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Accuracy: 0.7622\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Precision: 0.7629\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "Recall: 0.7622\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "F1 Score: 0.7488\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "        Fake       0.76      0.91      0.83       388\n",
      "        Real       0.77      0.51      0.61       226\n",
      "\n",
      "    accuracy                           0.76       614\n",
      "   macro avg       0.76      0.71      0.72       614\n",
      "weighted avg       0.76      0.76      0.75       614\n",
      "\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_ros, y_ros, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "9cb9229a-ec70-4467-8b3c-3b74c352bcf5",
   "metadata": {},
   "source": [
    "## SMOTE"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "id": "15ebce93-e35d-469c-bdac-8c98bc9bb750",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After SMOTE: Counter({'Fake': 903, 'Real': 903})\n"
     ]
    }
   ],
   "source": [
    "from imblearn.over_sampling import SMOTE\n",
    "smote = SMOTE(random_state=42)\n",
    "X_smote, y_smote = smote.fit_resample(X_train_vec, y_train)\n",
    "print(\"After SMOTE:\", Counter(y_smote))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 16,
   "id": "df857aa5-b59f-4756-a6b9-1fed1e82b6f7",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 100}\n",
      "Accuracy: 0.7394\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 100}\n",
      "Precision: 0.7336\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 100}\n",
      "Recall: 0.7394\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 100}\n",
      "F1 Score: 0.7302\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 100}\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "        Fake       0.76      0.86      0.81       388\n",
      "        Real       0.69      0.53      0.60       226\n",
      "\n",
      "    accuracy                           0.74       614\n",
      "   macro avg       0.72      0.69      0.70       614\n",
      "weighted avg       0.73      0.74      0.73       614\n",
      "\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_smote, y_smote, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "c78dce52-5295-4007-a4d2-dac2dbedb0f5",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python (NLP2)",
   "language": "python",
   "name": "nlp2"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.11.13"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
