{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "a1ae1df5-3ef6-4fa0-8ca0-e422696e098c",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 1. Import libraries \n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "import re\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, classification_report\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer\n",
    "from imblearn.over_sampling import RandomOverSampler, SMOTE\n",
    "from imblearn.under_sampling import RandomUnderSampler\n",
    "from collections import Counter"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "3f0d7402-0bc5-4552-aee2-e0f8932bbdca",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 2. Load dataset \n",
    "df = pd.read_csv(\"sentimentdataset.csv\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "5fb2df61-f38c-49ef-8f31-c6c6ee3f03b0",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 3. Map sentiments \n",
    "positive_labels = [\n",
    "    'positive', 'joy', 'excitement', 'contentment', 'gratitude', 'happy',\n",
    "    'hopeful', 'awe', 'acceptance', 'euphoria', 'radiance', 'serenity',\n",
    "    'artisticburst', 'harmony', 'elegance', 'confidence', 'pride', 'peace',\n",
    "    'admiration', 'love', 'relief', 'amusement', 'enlightenment', 'courage',\n",
    "    'satisfaction', 'cheerful', 'delight', 'wonder', 'optimism'\n",
    "]\n",
    "negative_labels = [\n",
    "    'negative', 'despair', 'grief', 'sad', 'loneliness', 'embarrassed',\n",
    "    'confusion', 'fear', 'anger', 'rage', 'anxiety', 'heartbreak', 'frustration',\n",
    "    'disgust', 'disappointment', 'resentment', 'envy', 'guilt', 'worry',\n",
    "    'doubt', 'melancholy', 'regret', 'shame', 'boredom', 'rejection', 'insecurity'\n",
    "]\n",
    "\n",
    "def map_sentiment(label):\n",
    "    label = str(label).strip().lower()\n",
    "    if label in positive_labels:\n",
    "        return \"positive\"\n",
    "    elif label in negative_labels:\n",
    "        return \"negative\"\n",
    "    else:\n",
    "        return None\n",
    "\n",
    "df['sentiment_bin'] = df['Sentiment'].apply(map_sentiment)\n",
    "df = df[df['sentiment_bin'].isin(['positive', 'negative'])].copy()\n",
    "df = df[['Text', 'sentiment_bin']].rename(columns={'Text':'text', 'sentiment_bin':'sentiment'})  # standardize names"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "601f37c6-a774-4059-9397-8f0a3a82ffa3",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 4. Preprocess text \n",
    "def preprocess_text(text):\n",
    "    text = str(text).lower()\n",
    "    text = re.sub(r'http\\S+|www.\\S+', '', text)\n",
    "    text = re.sub(r'<.*?>', '', text)\n",
    "    text = re.sub(r'[^a-zA-Z\\s]', '', text)\n",
    "    return text\n",
    "\n",
    "df[\"clean_text\"] = df[\"text\"].apply(preprocess_text)\n",
    "df = df.dropna()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "id": "ca71e64f-b851-429a-86e3-daf5e2acb9f4",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 5. Split into X/y\n",
    "X = df['clean_text']\n",
    "y = df['sentiment']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 11,
   "id": "50d86688-5860-4f6b-9ba9-686296440e7e",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 6. Train/test split\n",
    "X_train, X_test, y_train, y_test = train_test_split(\n",
    "    X, y, test_size=0.30, random_state=42, stratify=y)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "id": "6ff71eb4-a9ec-484c-a95f-fe7bef6b24d5",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 7. TF-IDF vectorization WITH ENGLISH STOPWORDS REMOVAL\n",
    "vectorizer = TfidfVectorizer(stop_words='english')\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec = vectorizer.transform(X_test)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "id": "eaf472d3-2bb5-48ec-994c-9e04029434d5",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 8. evaluation function\n",
    "def evaluate_model(model_name, y_true, y_pred):\n",
    "    accuracy = accuracy_score(y_true, y_pred)\n",
    "    precision = precision_score(y_true, y_pred, average='weighted')\n",
    "    recall = recall_score(y_true, y_pred, average='weighted')\n",
    "    f1 = f1_score(y_true, y_pred, average='weighted')\n",
    "    cm = confusion_matrix(y_true, y_pred)\n",
    "    report = classification_report(y_true, y_pred)\n",
    "    print(f\"\\nModel Name: {model_name}\")\n",
    "    print(f\"Accuracy: {accuracy:.4f}\")\n",
    "    print(f\"Precision: {precision:.4f}\")\n",
    "    print(f\"Recall: {recall:.4f}\")\n",
    "    print(f\"F1 Score: {f1:.4f}\")\n",
    "    print(\"Confusion Matrix:\")\n",
    "    print(cm)\n",
    "    print(report)\n",
    "    return {\n",
    "        'Model Name': model_name,\n",
    "        'Accuracy': accuracy,\n",
    "        'Precision': precision,\n",
    "        'Recall': recall,\n",
    "        'F1 Score': f1,\n",
    "        'Classification Report': report\n",
    "    }"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 17,
   "id": "fa47133f-c060-4597-9d9d-59827cbe5c6b",
   "metadata": {},
   "outputs": [],
   "source": [
    "# 9. Random Forest with GridSearchCV function\n",
    "def train_random_forest_classifier_with_grid_search(X_train_vec, y_train, X_test_vec, y_test, evaluate_model_func):\n",
    "    rf_classifier = RandomForestClassifier(random_state=42)\n",
    "    param_grid = {\n",
    "        'n_estimators': [100, 200, 300],\n",
    "        'max_depth': [None, 10, 20, 30],\n",
    "    }\n",
    "    grid_search = GridSearchCV(estimator=rf_classifier, param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)\n",
    "    grid_search.fit(X_train_vec, y_train)\n",
    "    best_rf_classifier = grid_search.best_estimator_\n",
    "    y_pred = best_rf_classifier.predict(X_test_vec)\n",
    "    evaluation_results = evaluate_model_func('RandomForestClassifier', y_test, y_pred)\n",
    "    print(\"\\nBest hyperparameters found by GridSearchCV:\")\n",
    "    print(grid_search.best_params_)\n",
    "    return evaluation_results"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "accf3446-e23d-4d15-8afd-0a54d6595ccc",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python [conda env:base] *",
   "language": "python",
   "name": "conda-base-py"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.7"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
