# -*- coding: utf-8 -*-
"""lec9.ipynb

Automatically generated by Colab.

Original file is located at
    https://colab.research.google.com/drive/1-weKiAmZJ8Bq5QcpWY9QZ187nDQ4N2Wl
"""

# ================================================================
# 1) Upload Dataset
# ================================================================
from google.colab import files
import pandas as pd

print("📤 Upload unbalanceddataset.csv")
uploaded = files.upload()

df = pd.read_csv("unbalanceddataset.csv")
print("Dataset loaded successfully!")
df.head()


# ================================================================
# 2) Import NLP + ML libraries
# ================================================================
import re
import nltk
from nltk.corpus import stopwords
from nltk.tokenize import word_tokenize
from nltk.stem import PorterStemmer

# Download necessary NLTK packages
nltk.download("punkt")
nltk.download("punkt_tab")   # <-- NEW FIX
nltk.download("stopwords")


stop_words = set(stopwords.words("english"))
stemmer = PorterStemmer()


# ================================================================
# 3) Text Preprocessing Function
# (Cleaning → Tokenizing → Removing Stopwords → Stemming)
# ================================================================
def preprocess_text(text):
    """
    Clean and normalize English text:
    1. Lowercase
    2. Remove URLs
    3. Remove HTML tags
    4. Remove special characters / numbers
    5. Tokenization
    6. Remove stopwords
    7. Stemming
    """
    text = text.lower()
    text = re.sub(r"http\S+|www.\S+", "", text)
    text = re.sub(r"<.*?>", "", text)
    text = re.sub(r"[^a-zA-Z\s]", "", text)

    tokens = word_tokenize(text)
    tokens = [w for w in tokens if w not in stop_words]
    tokens = [stemmer.stem(w) for w in tokens]

    return " ".join(tokens)


# Apply preprocess
df["clean_text"] = df["text"].apply(preprocess_text)
df = df.dropna()
df.head()


# ================================================================
# 4) Train/Test Split + TF-IDF Vectorization
# ================================================================
from sklearn.model_selection import train_test_split
from sklearn.feature_extraction.text import TfidfVectorizer

X = df["clean_text"]
y = df["sentiment"]

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.30, random_state=42, stratify=y)

vectorizer = TfidfVectorizer(max_features=5000)
X_train_vec = vectorizer.fit_transform(X_train)
X_test_vec = vectorizer.transform(X_test)


# ================================================================
# 5) Evaluation Function (Common for all ML Models)
# ================================================================
from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
    classification_report,
    confusion_matrix
)

def evaluate_model(model_name, y_true, y_pred):
    """
    Print accuracy, precision, recall, f1-score, and full classification report.
    """
    print("\n======================================")
    print(f"MODEL: {model_name}")
    print("======================================")

    print(f"Accuracy:  {accuracy_score(y_true, y_pred):.4f}")
    print(f"Precision: {precision_score(y_true, y_pred, average='weighted'):.4f}")
    print(f"Recall:    {recall_score(y_true, y_pred, average='weighted'):.4f}")
    print(f"F1 Score:  {f1_score(y_true, y_pred, average='weighted'):.4f}")

    print("\nClassification Report:")
    print(classification_report(y_true, y_pred))

    print("Confusion Matrix:")
    print(confusion_matrix(y_true, y_pred))


# ================================================================
# 6) Logistic Regression
# ================================================================
from sklearn.linear_model import LogisticRegression

logreg = LogisticRegression(class_weight="balanced", max_iter=500)
logreg.fit(X_train_vec, y_train)
y_pred_log = logreg.predict(X_test_vec)

evaluate_model("Logistic Regression", y_test, y_pred_log)


# ================================================================
# 7) Linear SVM (Excellent for NLP)
# ================================================================
from sklearn.svm import LinearSVC

svm = LinearSVC(class_weight="balanced")
svm.fit(X_train_vec, y_train)
y_pred_svm = svm.predict(X_test_vec)

evaluate_model("Linear SVM", y_test, y_pred_svm)


# ================================================================
# 8) Naive Bayes (Classic for TF-IDF)
# ================================================================
from sklearn.naive_bayes import MultinomialNB

nb = MultinomialNB()
nb.fit(X_train_vec, y_train)
y_pred_nb = nb.predict(X_test_vec)

evaluate_model("Naive Bayes", y_test, y_pred_nb)


# ================================================================
# 9) k-Nearest Neighbors
# ================================================================
from sklearn.neighbors import KNeighborsClassifier

knn = KNeighborsClassifier(n_neighbors=7)
knn.fit(X_train_vec.toarray(), y_train)
y_pred_knn = knn.predict(X_test_vec.toarray())

evaluate_model("KNN", y_test, y_pred_knn)


# ================================================================
# 10) Random Forest + Grid Search (Optimized)
# ================================================================
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import GridSearchCV

def train_rf_with_grid(X_train_vec, y_train, X_test_vec, y_test):
    """
    Train Random Forest using Grid Search to find best hyperparameters.
    """
    rf = RandomForestClassifier(random_state=42)

    param_grid = {
        "n_estimators": [100, 200, 300],
        "max_depth": [None, 10, 20, 30]
    }

    grid = GridSearchCV(rf, param_grid, cv=5, n_jobs=-1, scoring="accuracy")
    grid.fit(X_train_vec, y_train)

    best_model = grid.best_estimator_
    y_pred = best_model.predict(X_test_vec)

    evaluate_model("Random Forest (GridSearchCV)", y_test, y_pred)
    print("Best Hyperparameters:", grid.best_params_)


print("\n===== BASELINE Random Forest =====")
train_rf_with_grid(X_train_vec, y_train, X_test_vec, y_test)


# ================================================================
# 11) Oversampling (RandomOverSampler)
# ================================================================
from imblearn.over_sampling import RandomOverSampler
from collections import Counter

ros = RandomOverSampler(random_state=42)
X_ros, y_ros = ros.fit_resample(X_train_vec, y_train)
print("\nAfter Oversampling:", Counter(y_ros))

train_rf_with_grid(X_ros, y_ros, X_test_vec, y_test)


# ================================================================
# 12) Undersampling (RandomUnderSampler)
# ================================================================
from imblearn.under_sampling import RandomUnderSampler

rus = RandomUnderSampler(random_state=42)
X_rus, y_rus = rus.fit_resample(X_train_vec, y_train)
print("\nAfter Undersampling:", Counter(y_rus))

train_rf_with_grid(X_rus, y_rus, X_test_vec, y_test)


# ================================================================
# 13) SMOTE (Synthetic Minority Oversampling)
# ================================================================
from imblearn.over_sampling import SMOTE

smote = SMOTE(random_state=42)
X_sm, y_sm = smote.fit_resample(X_train_vec, y_train)
print("\nAfter SMOTE:", Counter(y_sm))

train_rf_with_grid(X_sm, y_sm, X_test_vec, y_test)