{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "61c4192f-685a-4ebb-a756-f1c6bc3eb24e",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "                                                text sentiment\n",
      "0  Java Concurrency in Practice is probably the b...  positive\n",
      "1    haha aww hun i bet you are more creative tha...  positive\n",
      "2  _pickle lol, thank you very much Hope you`re h...  positive\n",
      "3  Out for an evening on the town with jeremy. Sa...  negative\n",
      "4   - just took over the #1 Most Endorsed spot on...  positive\n",
      "=== Logistic Regression Results ===\n",
      "Accuracy: 0.8134615384615385\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       0.90      0.22      0.35       120\n",
      "    positive       0.81      0.99      0.89       400\n",
      "\n",
      "    accuracy                           0.81       520\n",
      "   macro avg       0.85      0.60      0.62       520\n",
      "weighted avg       0.83      0.81      0.77       520\n",
      "\n",
      "=== Naive Bayes Results ===\n",
      "Accuracy: 0.7711538461538462\n",
      "              precision    recall  f1-score   support\n",
      "\n",
      "    negative       1.00      0.01      0.02       120\n",
      "    positive       0.77      1.00      0.87       400\n",
      "\n",
      "    accuracy                           0.77       520\n",
      "   macro avg       0.89      0.50      0.44       520\n",
      "weighted avg       0.82      0.77      0.67       520\n",
      "\n"
     ]
    }
   ],
   "source": [
    "# Step 1: Import necessary libraries\n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "import re\n",
    "import string\n",
    "\n",
    "from sklearn.model_selection import train_test_split\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.naive_bayes import MultinomialNB\n",
    "from sklearn.metrics import classification_report, accuracy_score\n",
    "\n",
    "# Step 2: Load your dataset\n",
    "file_path = r\"E:\\Downloads\\unbalanceddataset.csv\"  # <-- your path\n",
    "data = pd.read_csv(file_path)\n",
    "\n",
    "# Display the first few rows\n",
    "print(data.head())\n",
    "\n",
    "# Step 3: Data Preprocessing\n",
    "def clean_text(text):\n",
    "    # Lowercase\n",
    "    text = text.lower()\n",
    "    # Remove punctuation\n",
    "    text = text.translate(str.maketrans('', '', string.punctuation))\n",
    "    # Remove numbers\n",
    "    text = re.sub(r'\\d+', '', text)\n",
    "    # Remove extra whitespace\n",
    "    text = text.strip()\n",
    "    return text\n",
    "\n",
    "# Apply cleaning\n",
    "data['text'] = data['text'].apply(clean_text)\n",
    "\n",
    "# Step 4: Split dataset\n",
    "X = data['text']            # Features\n",
    "y = data['sentiment']       # Target\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(\n",
    "    X, y, test_size=0.2, random_state=42, stratify=y\n",
    ")\n",
    "\n",
    "# Step 5: Feature Representation (TF-IDF)\n",
    "vectorizer = TfidfVectorizer(max_features=5000)\n",
    "X_train_tfidf = vectorizer.fit_transform(X_train)\n",
    "X_test_tfidf = vectorizer.transform(X_test)\n",
    "\n",
    "# Step 6: Train Models\n",
    "\n",
    "# 6.1 Logistic Regression\n",
    "lr_model = LogisticRegression()\n",
    "lr_model.fit(X_train_tfidf, y_train)\n",
    "y_pred_lr = lr_model.predict(X_test_tfidf)\n",
    "\n",
    "print(\"=== Logistic Regression Results ===\")\n",
    "print(\"Accuracy:\", accuracy_score(y_test, y_pred_lr))\n",
    "print(classification_report(y_test, y_pred_lr))\n",
    "\n",
    "# 6.2 Naive Bayes\n",
    "nb_model = MultinomialNB()\n",
    "nb_model.fit(X_train_tfidf, y_train)\n",
    "y_pred_nb = nb_model.predict(X_test_tfidf)\n",
    "\n",
    "print(\"=== Naive Bayes Results ===\")\n",
    "print(\"Accuracy:\", accuracy_score(y_test, y_pred_nb))\n",
    "print(classification_report(y_test, y_pred_nb))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "f3f752a6-b287-42d5-ab94-b9e4c306fa74",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python [conda env:base] *",
   "language": "python",
   "name": "conda-base-py"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.7"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
