{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "5399af44-fd66-40b8-a9b4-45baba484394",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "\n",
      "=== RandomForest (Baseline (no resampling)) ===\n",
      "Accuracy: 0.8513\n",
      "Precision(weighted): 0.8518\n",
      "Recall(weighted): 0.8513\n",
      "F1(weighted): 0.8317\n",
      "\n",
      "Classification Report:\n",
      "               precision    recall  f1-score   support\n",
      "\n",
      "    negative     0.8556    0.4278    0.5704       180\n",
      "    positive     0.8507    0.9783    0.9101       600\n",
      "\n",
      "    accuracy                         0.8513       780\n",
      "   macro avg     0.8531    0.7031    0.7402       780\n",
      "weighted avg     0.8518    0.8513    0.8317       780\n",
      "\n",
      "Best Params: {'max_depth': None, 'n_estimators': 300}\n",
      "\n",
      "=== RandomForest (RandomOverSampler) ===\n",
      "Accuracy: 0.8654\n",
      "Precision(weighted): 0.8610\n",
      "Recall(weighted): 0.8654\n",
      "F1(weighted): 0.8624\n",
      "\n",
      "Classification Report:\n",
      "               precision    recall  f1-score   support\n",
      "\n",
      "    negative     0.7358    0.6500    0.6903       180\n",
      "    positive     0.8986    0.9300    0.9140       600\n",
      "\n",
      "    accuracy                         0.8654       780\n",
      "   macro avg     0.8172    0.7900    0.8021       780\n",
      "weighted avg     0.8610    0.8654    0.8624       780\n",
      "\n",
      "Best Params: {'max_depth': None, 'n_estimators': 300}\n",
      "\n",
      "=== RandomForest (RandomUnderSampler) ===\n",
      "Accuracy: 0.7372\n",
      "Precision(weighted): 0.8376\n",
      "Recall(weighted): 0.7372\n",
      "F1(weighted): 0.7574\n",
      "\n",
      "Classification Report:\n",
      "               precision    recall  f1-score   support\n",
      "\n",
      "    negative     0.4633    0.8778    0.6065       180\n",
      "    positive     0.9499    0.6950    0.8027       600\n",
      "\n",
      "    accuracy                         0.7372       780\n",
      "   macro avg     0.7066    0.7864    0.7046       780\n",
      "weighted avg     0.8376    0.7372    0.7574       780\n",
      "\n",
      "Best Params: {'max_depth': 30, 'n_estimators': 200}\n",
      "\n",
      "=== RandomForest (SMOTE) ===\n",
      "Accuracy: 0.8526\n",
      "Precision(weighted): 0.8464\n",
      "Recall(weighted): 0.8526\n",
      "F1(weighted): 0.8391\n",
      "\n",
      "Classification Report:\n",
      "               precision    recall  f1-score   support\n",
      "\n",
      "    negative     0.7928    0.4889    0.6048       180\n",
      "    positive     0.8625    0.9617    0.9094       600\n",
      "\n",
      "    accuracy                         0.8526       780\n",
      "   macro avg     0.8276    0.7253    0.7571       780\n",
      "weighted avg     0.8464    0.8526    0.8391       780\n",
      "\n",
      "Best Params: {'max_depth': None, 'n_estimators': 200}\n",
      "\n",
      "\n",
      "=== SUMMARY (JSON) ===\n",
      "{\n",
      "  \"baseline\": {\n",
      "    \"metrics\": {\n",
      "      \"model\": \"RandomForest (Baseline (no resampling))\",\n",
      "      \"accuracy\": 0.8512820512820513,\n",
      "      \"precision_weighted\": 0.8518394648829433,\n",
      "      \"recall_weighted\": 0.8512820512820513,\n",
      "      \"f1_weighted\": 0.8316835619161201,\n",
      "      \"report\": \"              precision    recall  f1-score   support\\n\\n    negative     0.8556    0.4278    0.5704       180\\n    positive     0.8507    0.9783    0.9101       600\\n\\n    accuracy                         0.8513       780\\n   macro avg     0.8531    0.7031    0.7402       780\\nweighted avg     0.8518    0.8513    0.8317       780\\n\"\n",
      "    },\n",
      "    \"best_params\": {\n",
      "      \"max_depth\": null,\n",
      "      \"n_estimators\": 300\n",
      "    }\n",
      "  },\n",
      "  \"oversample\": {\n",
      "    \"metrics\": {\n",
      "      \"model\": \"RandomForest (RandomOverSampler)\",\n",
      "      \"accuracy\": 0.8653846153846154,\n",
      "      \"precision_weighted\": 0.8610041858606255,\n",
      "      \"recall_weighted\": 0.8653846153846154,\n",
      "      \"f1_weighted\": 0.8623727384789331,\n",
      "      \"report\": \"              precision    recall  f1-score   support\\n\\n    negative     0.7358    0.6500    0.6903       180\\n    positive     0.8986    0.9300    0.9140       600\\n\\n    accuracy                         0.8654       780\\n   macro avg     0.8172    0.7900    0.8021       780\\nweighted avg     0.8610    0.8654    0.8624       780\\n\"\n",
      "    },\n",
      "    \"best_params\": {\n",
      "      \"max_depth\": null,\n",
      "      \"n_estimators\": 300\n",
      "    }\n",
      "  },\n",
      "  \"undersample\": {\n",
      "    \"metrics\": {\n",
      "      \"model\": \"RandomForest (RandomUnderSampler)\",\n",
      "      \"accuracy\": 0.7371794871794872,\n",
      "      \"precision_weighted\": 0.8376069517960913,\n",
      "      \"recall_weighted\": 0.7371794871794872,\n",
      "      \"f1_weighted\": 0.7574251326567427,\n",
      "      \"report\": \"              precision    recall  f1-score   support\\n\\n    negative     0.4633    0.8778    0.6065       180\\n    positive     0.9499    0.6950    0.8027       600\\n\\n    accuracy                         0.7372       780\\n   macro avg     0.7066    0.7864    0.7046       780\\nweighted avg     0.8376    0.7372    0.7574       780\\n\"\n",
      "    },\n",
      "    \"best_params\": {\n",
      "      \"max_depth\": 30,\n",
      "      \"n_estimators\": 200\n",
      "    }\n",
      "  },\n",
      "  \"smote\": {\n",
      "    \"metrics\": {\n",
      "      \"model\": \"RandomForest (SMOTE)\",\n",
      "      \"accuracy\": 0.8525641025641025,\n",
      "      \"precision_weighted\": 0.8463993486415011,\n",
      "      \"recall_weighted\": 0.8525641025641025,\n",
      "      \"f1_weighted\": 0.8390928934907878,\n",
      "      \"report\": \"              precision    recall  f1-score   support\\n\\n    negative     0.7928    0.4889    0.6048       180\\n    positive     0.8625    0.9617    0.9094       600\\n\\n    accuracy                         0.8526       780\\n   macro avg     0.8276    0.7253    0.7571       780\\nweighted avg     0.8464    0.8526    0.8391       780\\n\"\n",
      "    },\n",
      "    \"best_params\": {\n",
      "      \"max_depth\": null,\n",
      "      \"n_estimators\": 200\n",
      "    }\n",
      "  }\n",
      "}\n"
     ]
    }
   ],
   "source": [
    "\n",
    "import re\n",
    "import json\n",
    "import numpy as np\n",
    "import pandas as pd\n",
    "\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.metrics import (accuracy_score, precision_score, recall_score,\n",
    "                             f1_score, classification_report, confusion_matrix)\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "\n",
    "import nltk\n",
    "from nltk.corpus import stopwords\n",
    "from nltk.tokenize import word_tokenize\n",
    "from nltk.stem import PorterStemmer\n",
    "\n",
    "from imblearn.over_sampling import RandomOverSampler\n",
    "from imblearn.under_sampling import RandomUnderSampler\n",
    "from imblearn.over_sampling import SMOTE\n",
    "\n",
    "try:\n",
    "    nltk.data.find(\"tokenizers/punkt\")\n",
    "except LookupError:\n",
    "    nltk.download(\"punkt\")\n",
    "try:\n",
    "    nltk.data.find(\"corpora/stopwords\")\n",
    "except LookupError:\n",
    "    nltk.download(\"stopwords\")\n",
    "\n",
    "df = pd.read_csv(\"unbalanceddataset.csv\")\n",
    "\n",
    "stemmer = PorterStemmer()\n",
    "stop_words = set(stopwords.words(\"english\"))\n",
    "\n",
    "def preprocess_text(text: str) -> str:\n",
    "    if not isinstance(text, str):\n",
    "        text = str(text)\n",
    "\n",
    "    text = text.lower()\n",
    "    text = re.sub(r\"http\\S+|www\\.\\S+\", \"\", text)         \n",
    "    text = re.sub(r\"<.*?>\", \"\", text)                    \n",
    "    text = re.sub(r\"[^a-zA-Z\\s]\", \"\", text)             \n",
    "    tokens = word_tokenize(text)\n",
    "    tokens = [t for t in tokens if t not in stop_words and len(t) > 1]\n",
    "    tokens = [stemmer.stem(t) for t in tokens]\n",
    "    return \" \".join(tokens)\n",
    "\n",
    "df[\"clean_text\"] = df[\"text\"].apply(preprocess_text)\n",
    "df = df.dropna(subset=[\"clean_text\", \"sentiment\"])\n",
    "\n",
    "\n",
    "X = df[\"clean_text\"]\n",
    "y = df[\"sentiment\"]\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(\n",
    "    X, y, test_size=0.30, random_state=42, stratify=y\n",
    ")\n",
    "\n",
    "\n",
    "vectorizer = TfidfVectorizer()\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec  = vectorizer.transform(X_test)\n",
    "\n",
    "\n",
    "def evaluate_model(model_name, y_true, y_pred):\n",
    "    metrics = {\n",
    "        \"model\": model_name,\n",
    "        \"accuracy\": accuracy_score(y_true, y_pred),\n",
    "        \"precision_weighted\": precision_score(y_true, y_pred, average=\"weighted\"),\n",
    "        \"recall_weighted\": recall_score(y_true, y_pred, average=\"weighted\"),\n",
    "        \"f1_weighted\": f1_score(y_true, y_pred, average=\"weighted\"),\n",
    "        \"report\": classification_report(y_true, y_pred, digits=4)\n",
    "    }\n",
    "    return metrics\n",
    "\n",
    "def print_metrics(m):\n",
    "    print(f\"\\n=== {m['model']} ===\")\n",
    "    print(f\"Accuracy: {m['accuracy']:.4f}\")\n",
    "    print(f\"Precision(weighted): {m['precision_weighted']:.4f}\")\n",
    "    print(f\"Recall(weighted): {m['recall_weighted']:.4f}\")\n",
    "    print(f\"F1(weighted): {m['f1_weighted']:.4f}\")\n",
    "    print(\"\\nClassification Report:\\n\", m[\"report\"])\n",
    "\n",
    "\n",
    "def train_random_forest_with_grid(Xtr, ytr, Xte, yte, label):\n",
    "    rf = RandomForestClassifier(random_state=42)\n",
    "    param_grid = {\n",
    "        \"n_estimators\": [100, 200, 300],\n",
    "        \"max_depth\": [None, 10, 20, 30],\n",
    "    }\n",
    "    gs = GridSearchCV(rf, param_grid=param_grid, cv=5, scoring=\"accuracy\", n_jobs=-1)\n",
    "    gs.fit(Xtr, ytr)\n",
    "    best = gs.best_estimator_\n",
    "    y_pred = best.predict(Xte)\n",
    "    metrics = evaluate_model(f\"RandomForest ({label})\", yte, y_pred)\n",
    "    print_metrics(metrics)\n",
    "    print(\"Best Params:\", gs.best_params_)\n",
    "    return metrics, gs.best_params_\n",
    "\n",
    "\n",
    "\n",
    "baseline_metrics, baseline_params = train_random_forest_with_grid(\n",
    "    X_train_vec, y_train, X_test_vec, y_test, label=\"Baseline (no resampling)\"\n",
    ")\n",
    "\n",
    "\n",
    "ros = RandomOverSampler(random_state=42)\n",
    "X_ros, y_ros = ros.fit_resample(X_train_vec, y_train)\n",
    "\n",
    "ros_metrics, ros_params = train_random_forest_with_grid(\n",
    "    X_ros, y_ros, X_test_vec, y_test, label=\"RandomOverSampler\"\n",
    ")\n",
    "\n",
    "\n",
    "rus = RandomUnderSampler(random_state=42)\n",
    "X_rus, y_rus = rus.fit_resample(X_train_vec, y_train)\n",
    "\n",
    "rus_metrics, rus_params = train_random_forest_with_grid(\n",
    "    X_rus, y_rus, X_test_vec, y_test, label=\"RandomUnderSampler\"\n",
    ")\n",
    "\n",
    "smote = SMOTE(random_state=42)\n",
    "X_sm, y_sm = smote.fit_resample(X_train_vec, y_train)\n",
    "\n",
    "smote_metrics, smote_params = train_random_forest_with_grid(\n",
    "    X_sm, y_sm, X_test_vec, y_test, label=\"SMOTE\"\n",
    ")\n",
    "\n",
    "\n",
    "summary = {\n",
    "    \"baseline\": {\"metrics\": baseline_metrics, \"best_params\": baseline_params},\n",
    "    \"oversample\": {\"metrics\": ros_metrics, \"best_params\": ros_params},\n",
    "    \"undersample\": {\"metrics\": rus_metrics, \"best_params\": rus_params},\n",
    "    \"smote\": {\"metrics\": smote_metrics, \"best_params\": smote_params},\n",
    "}\n",
    "\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "ab8453bc-5de4-464b-b869-3221f88726af",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python (env)",
   "language": "python",
   "name": "env"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.5"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
