{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "19a5b9ae-a3c2-4a07-95bc-4b1c8cdf8b0d",
   "metadata": {},
   "outputs": [
    {
     "name": "stderr",
     "output_type": "stream",
     "text": [
      "[nltk_data] Downloading package punkt to /Users/meshmoh/nltk_data...\n",
      "[nltk_data]   Package punkt is already up-to-date!\n",
      "[nltk_data] Downloading package punkt_tab to\n",
      "[nltk_data]     /Users/meshmoh/nltk_data...\n",
      "[nltk_data]   Package punkt_tab is already up-to-date!\n",
      "[nltk_data] Downloading package stopwords to\n",
      "[nltk_data]     /Users/meshmoh/nltk_data...\n",
      "[nltk_data]   Package stopwords is already up-to-date!\n",
      "[nltk_data] Downloading package wordnet to /Users/meshmoh/nltk_data...\n",
      "[nltk_data]   Package wordnet is already up-to-date!\n"
     ]
    },
    {
     "data": {
      "text/plain": [
       "True"
      ]
     },
     "execution_count": 2,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "import pandas as pd\n",
    "import numpy as np\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.preprocessing import StandardScaler, LabelEncoder\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.tree import DecisionTreeClassifier\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.svm import SVC\n",
    "from sklearn.naive_bayes import GaussianNB\n",
    "from sklearn.neighbors import KNeighborsClassifier\n",
    "from sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, classification_report\n",
    "import matplotlib.pyplot as plt\n",
    "import seaborn as sns\n",
    "from sklearn.model_selection import GridSearchCV\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.metrics import classification_report\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer\n",
    "from sklearn.model_selection import train_test_split\n",
    "from sklearn.feature_extraction.text import CountVectorizer\n",
    "\n",
    "import re\n",
    "import nltk\n",
    "\n",
    "from nltk.corpus import stopwords\n",
    "from nltk.tokenize import word_tokenize\n",
    "from nltk.stem import PorterStemmer\n",
    "from nltk.stem import WordNetLemmatizer\n",
    "\n",
    "# Download NLTK resources (only first time)\n",
    "nltk.download('punkt')\n",
    "nltk.download('punkt_tab')\n",
    "nltk.download('stopwords')\n",
    "nltk.download('wordnet')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "18ba216e-4cbb-48ef-bf0f-4bbe6762ca48",
   "metadata": {},
   "outputs": [],
   "source": [
    "def evaluate_model(model_name, y_true, y_pred):\n",
    "    \n",
    "    # Calculate metrics\n",
    "    accuracy = accuracy_score(y_true, y_pred)\n",
    "    precision = precision_score(y_true, y_pred, average='weighted') \n",
    "    recall = recall_score(y_true, y_pred, average='weighted')\n",
    "    f1 = f1_score(y_true, y_pred, average='weighted')\n",
    "    cm = confusion_matrix(y_true, y_pred)\n",
    "    \n",
    "    # Create a report\n",
    "    report = classification_report(y_true, y_pred)\n",
    "    \n",
    "    # Output results\n",
    "    metrics = {\n",
    "        'Model Name': model_name,\n",
    "        'Accuracy': accuracy,\n",
    "        'Precision': precision,\n",
    "        'Recall': recall,\n",
    "        'F1 Score': f1,\n",
    "        'Classification Report': report\n",
    "    }\n",
    "    return metrics"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "id": "03a86ed8-81eb-4a76-8915-c2d212153c51",
   "metadata": {},
   "outputs": [],
   "source": [
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.model_selection import GridSearchCV\n",
    "\n",
    "def train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model_func):\n",
    "    # Define the Random Forest Classifier\n",
    "    rf_classifier = RandomForestClassifier(random_state=42)\n",
    "    # Define the hyperparameters for grid search\n",
    "    param_grid = {\n",
    "        'n_estimators': [100, 200, 300],\n",
    "        'max_depth': [None, 10, 20, 30],\n",
    "    }\n",
    "    # Initialize GridSearchCV\n",
    "    grid_search = GridSearchCV(estimator=rf_classifier, param_grid=param_grid, cv=5,\n",
    "                               scoring='accuracy', n_jobs=-1)\n",
    "    # Fit the grid search to the data\n",
    "    grid_search.fit(X_train_vec, y_train)\n",
    "    # Get the best estimator from grid search\n",
    "    best_rf_classifier = grid_search.best_estimator_\n",
    "    # Make predictions using the best model\n",
    "    y_pred = best_rf_classifier.predict(X_test_vec)\n",
    "    # Evaluate the model\n",
    "    evaluation_results = evaluate_model_func('RandomForestClassifier', y_test, y_pred)\n",
    "    # Print the evaluation results\n",
    "    for key, value in evaluation_results.items():\n",
    "        if key == 'Classification Report':\n",
    "            print(value)  # Print report separately for better readability\n",
    "        else:\n",
    "            print(f\"{key}: {value:.4f}\" if isinstance(value, float) else f\"{key}: \\n{value}\")\n",
    "    # Print the best parameters found by grid search\n",
    "    print(\"\\nBest hyperparameters found by GridSearchCV:\")\n",
    "    print(grid_search.best_params_)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "432ccc43-718f-4407-9994-5349358607c1",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>text</th>\n",
       "      <th>sentiment</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>Java Concurrency in Practice is probably the b...</td>\n",
       "      <td>positive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>haha aww hun i bet you are more creative tha...</td>\n",
       "      <td>positive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>_pickle lol, thank you very much Hope you`re h...</td>\n",
       "      <td>positive</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>3</th>\n",
       "      <td>Out for an evening on the town with jeremy. Sa...</td>\n",
       "      <td>negative</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>4</th>\n",
       "      <td>- just took over the #1 Most Endorsed spot on...</td>\n",
       "      <td>positive</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "                                                text sentiment\n",
       "0  Java Concurrency in Practice is probably the b...  positive\n",
       "1    haha aww hun i bet you are more creative tha...  positive\n",
       "2  _pickle lol, thank you very much Hope you`re h...  positive\n",
       "3  Out for an evening on the town with jeremy. Sa...  negative\n",
       "4   - just took over the #1 Most Endorsed spot on...  positive"
      ]
     },
     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df = pd.read_csv(\"unbalanceddataset.csv\") \n",
    "df.head()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "id": "accdef64-61fb-4864-93c9-c7b94b91b136",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "sentiment\n",
       "positive    2000\n",
       "negative     600\n",
       "Name: count, dtype: int64"
      ]
     },
     "execution_count": 6,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df['sentiment'].value_counts()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "0196bcd3-a306-402e-b654-90f6c12941c7",
   "metadata": {},
   "outputs": [],
   "source": [
    "stemmer = PorterStemmer()\n",
    "stop_words = set(stopwords.words('english'))\n",
    "\n",
    "def preprocess_text(text):\n",
    "    text = text.lower()\n",
    "    # 2. Remove URLs\n",
    "    text = re.sub(r'http\\S+|www.\\S+', '', text)\n",
    "    # 3. Remove HTML tags\n",
    "    text = re.sub(r'<.*?>', '', text)\n",
    "    # 4. Remove Special Characters, Numbers, Punctuation\n",
    "    text = re.sub(r'[^a-zA-Z\\s]', '', text)\n",
    "    # 5. Tokenization\n",
    "    tokens = word_tokenize(text)\n",
    "    # 6. Remove Stop Words\n",
    "    tokens = [word for word in tokens if word not in stop_words]\n",
    "    # 7. Stemming\n",
    "    stemmed_tokens = [stemmer.stem(word) for word in tokens]\n",
    "    # 8. Final clean text reconstruction (optional)\n",
    "    clean_text = ' '.join(stemmed_tokens)\n",
    "    \n",
    "    return clean_text"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "id": "cbc53959-b0aa-4b36-9d3e-01216e7d2043",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
       "0       java concurr practic probabl best java book iv...\n",
       "1                                haha aww hun bet creativ\n",
       "2                pickl lol thank much hope your great day\n",
       "3                    even town jeremi sad carri cant come\n",
       "4               took endors spot twindexxcom thank endors\n",
       "                              ...                        \n",
       "2595    what yall made earli night think ima bout take...\n",
       "2596                                                 agre\n",
       "2597                                           yeah thank\n",
       "2598                                    good morn everyon\n",
       "2599           em babi start kindergarten crazi summer go\n",
       "Name: clean_text, Length: 2600, dtype: str"
      ]
     },
     "execution_count": 8,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df[\"clean_text\"] = df[\"text\"].apply(preprocess_text)\n",
    "df = df.dropna()\n",
    "df['clean_text']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "id": "99c363a1-9fcc-43be-a31d-a35813426f7b",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>text</th>\n",
       "      <th>sentiment</th>\n",
       "      <th>clean_text</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>Java Concurrency in Practice is probably the b...</td>\n",
       "      <td>positive</td>\n",
       "      <td>java concurr practic probabl best java book iv...</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>haha aww hun i bet you are more creative tha...</td>\n",
       "      <td>positive</td>\n",
       "      <td>haha aww hun bet creativ</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>_pickle lol, thank you very much Hope you`re h...</td>\n",
       "      <td>positive</td>\n",
       "      <td>pickl lol thank much hope your great day</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>3</th>\n",
       "      <td>Out for an evening on the town with jeremy. Sa...</td>\n",
       "      <td>negative</td>\n",
       "      <td>even town jeremi sad carri cant come</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>4</th>\n",
       "      <td>- just took over the #1 Most Endorsed spot on...</td>\n",
       "      <td>positive</td>\n",
       "      <td>took endors spot twindexxcom thank endors</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "                                                text sentiment  \\\n",
       "0  Java Concurrency in Practice is probably the b...  positive   \n",
       "1    haha aww hun i bet you are more creative tha...  positive   \n",
       "2  _pickle lol, thank you very much Hope you`re h...  positive   \n",
       "3  Out for an evening on the town with jeremy. Sa...  negative   \n",
       "4   - just took over the #1 Most Endorsed spot on...  positive   \n",
       "\n",
       "                                          clean_text  \n",
       "0  java concurr practic probabl best java book iv...  \n",
       "1                           haha aww hun bet creativ  \n",
       "2           pickl lol thank much hope your great day  \n",
       "3               even town jeremi sad carri cant come  \n",
       "4          took endors spot twindexxcom thank endors  "
      ]
     },
     "execution_count": 9,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df.head()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 10,
   "id": "cf78bfbb-307d-4ed8-b1bf-0ed737ddeb91",
   "metadata": {},
   "outputs": [],
   "source": [
    "X = df['clean_text']\n",
    "y = df['sentiment']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 11,
   "id": "c40a8e1d-6de2-4908-aca0-ffe9ae8d94a5",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=42, stratify=y)\n",
    "\n",
    "vectorizer = TfidfVectorizer()\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec = vectorizer.transform(X_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 12,
   "id": "12e70af2-1218-48c0-9541-26f76a0c40de",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.8462\n",
      "Precision: 0.8428\n",
      "Recall: 0.8462\n",
      "F1 Score: 0.8272\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       0.82      0.43      0.56       180\n",
      "    positive       0.85      0.97      0.91       600\n",
      "\n",
      "    accuracy                           0.85       780\n",
      "   macro avg       0.83      0.70      0.73       780\n",
      "weighted avg       0.84      0.85      0.83       780\n",
      "\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 200}\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "id": "30196a45-1726-4103-a9ca-f9f512d791ed",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=42, stratify=y)\n",
    "\n",
    "vectorizer = TfidfVectorizer()\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec = vectorizer.transform(X_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 14,
   "id": "10f9c641-90fa-4d08-aade-a32bc2d0be33",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After RandomOversampling: Counter({'positive': 1400, 'negative': 1400})\n"
     ]
    }
   ],
   "source": [
    "from imblearn.over_sampling import RandomOverSampler\n",
    "from collections import Counter\n",
    "\n",
    "ros = RandomOverSampler(random_state=42)\n",
    "X_ros, y_ros = ros.fit_resample(X_train_vec, y_train)\n",
    "print(\"After RandomOversampling:\", Counter(y_ros))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "id": "cbc3405d-0cc2-440b-a966-05c965219f83",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.8526\n",
      "Precision: 0.8455\n",
      "Recall: 0.8526\n",
      "F1 Score: 0.8467\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       0.72      0.58      0.65       180\n",
      "    positive       0.88      0.93      0.91       600\n",
      "\n",
      "    accuracy                           0.85       780\n",
      "   macro avg       0.80      0.76      0.78       780\n",
      "weighted avg       0.85      0.85      0.85       780\n",
      "\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 300}\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_ros, y_ros, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 16,
   "id": "1d8cd19d-7ef7-4265-8ba7-64046582a94c",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After undersampling: Counter({'negative': 420, 'positive': 420})\n"
     ]
    }
   ],
   "source": [
    "from imblearn.under_sampling import RandomUnderSampler\n",
    "\n",
    "rus = RandomUnderSampler(random_state=42)\n",
    "X_res, y_res = rus.fit_resample(X_train_vec, y_train)\n",
    "print(\"After undersampling:\", Counter(y_res))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 17,
   "id": "2983a4cc-1139-4025-9950-bff425a5aa3d",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.7231\n",
      "Precision: 0.8419\n",
      "Recall: 0.7231\n",
      "F1 Score: 0.7448\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       0.45      0.91      0.60       180\n",
      "    positive       0.96      0.67      0.79       600\n",
      "\n",
      "    accuracy                           0.72       780\n",
      "   macro avg       0.70      0.79      0.69       780\n",
      "weighted avg       0.84      0.72      0.74       780\n",
      "\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': 20, 'n_estimators': 200}\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_res, y_res, X_test_vec, y_test, evaluate_model)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 18,
   "id": "645842dc-c1bc-4754-b8c9-e93c14c12945",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "After SMOTE: Counter({'positive': 1400, 'negative': 1400})\n"
     ]
    }
   ],
   "source": [
    "from imblearn.over_sampling import SMOTE\n",
    "\n",
    "smote = SMOTE(random_state=42)\n",
    "X_smote, y_smote = smote.fit_resample(X_train_vec, y_train)\n",
    "print(\"After SMOTE:\", Counter(y_smote))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 19,
   "id": "0844d1b2-a28c-4c37-a914-7045db572b3a",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Model Name: \n",
      "RandomForestClassifier\n",
      "Accuracy: 0.8462\n",
      "Precision: 0.8379\n",
      "Recall: 0.8462\n",
      "F1 Score: 0.8329\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       0.76      0.48      0.59       180\n",
      "    positive       0.86      0.95      0.91       600\n",
      "\n",
      "    accuracy                           0.85       780\n",
      "   macro avg       0.81      0.72      0.75       780\n",
      "weighted avg       0.84      0.85      0.83       780\n",
      "\n",
      "\n",
      "Best hyperparameters found by GridSearchCV:\n",
      "{'max_depth': None, 'n_estimators': 200}\n"
     ]
    }
   ],
   "source": [
    "train_random_forest_classifier_with_grid_search(X_smote, y_smote, X_test_vec, y_test, evaluate_model)"
   ]
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3 (ipykernel)",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.13.15"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
