# Complete NLP Classification Pipeline in Python

import pandas as pd
import numpy as np
import re
import string

from sklearn.model_selection import train_test_split
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.linear_model import LogisticRegression
from sklearn.svm import LinearSVC
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix

# --------------------------------------------------
# 1. Load Dataset
# --------------------------------------------------
data_path = "/mnt/data/unbalanceddataset.csv"
df = pd.read_csv(data_path)

print("Dataset shape:", df.shape)
print(df.head())

# --------------------------------------------------
# 2. Data Preprocessing
# --------------------------------------------------
def clean_text(text):
    text = text.lower()
    text = re.sub(r"http\S+|www\S+", "", text)  # remove URLs
    text = re.sub(r"\d+", "", text)              # remove numbers
    text = text.translate(str.maketrans("", "", string.punctuation))
    text = text.strip()
    return text

df["clean_text"] = df["text"].apply(clean_text)

# Encode labels
df["label"] = df["sentiment"].map({"positive": 1, "negative": 0})

# --------------------------------------------------
# 3. Train / Test Split
# --------------------------------------------------
X_train, X_test, y_train, y_test = train_test_split(
    df["clean_text"],
    df["label"],
    test_size=0.2,
    random_state=42,
    stratify=df["label"]
)

print("\nTrain size:", X_train.shape)
print("Test size:", X_test.shape)

# --------------------------------------------------
# 4. Feature Representation (TF-IDF)
# --------------------------------------------------
tfidf = TfidfVectorizer(
    max_features=5000,
    ngram_range=(1, 2),
    stop_words="english"
)

X_train_tfidf = tfidf.fit_transform(X_train)
X_test_tfidf = tfidf.transform(X_test)

print("\nTF-IDF Train shape:", X_train_tfidf.shape)
print("TF-IDF Test shape:", X_test_tfidf.shape)

# --------------------------------------------------
# 5. Train Models
# --------------------------------------------------

# Model 1: Logistic Regression
lr_model = LogisticRegression(max_iter=1000, class_weight="balanced")
lr_model.fit(X_train_tfidf, y_train)

# Model 2: Linear SVM
svm_model = LinearSVC(class_weight="balanced")
svm_model.fit(X_train_tfidf, y_train)

# --------------------------------------------------
# 6. Evaluation
# --------------------------------------------------
def evaluate_model(model, X_test, y_test, model_name):
    y_pred = model.predict(X_test)
    print(f"\n===== {model_name} =====")
    print("Accuracy:", accuracy_score(y_test, y_pred))
    print("Confusion Matrix:\n", confusion_matrix(y_test, y_pred))
    print("Classification Report:\n", classification_report(y_test, y_pred))

evaluate_model(lr_model, X_test_tfidf, y_test, "Logistic Regression")
evaluate_model(svm_model, X_test_tfidf, y_test, "Linear SVM")

# --------------------------------------------------
# 7. Example Prediction
# --------------------------------------------------
sample_text = ["This product is absolutely amazing and I love it"]
sample_clean = [clean_text(sample_text[0])]
sample_vec = tfidf.transform(sample_clean)

print("\nSample Prediction (Logistic Regression):",
      "Positive" if lr_model.predict(sample_vec)[0] == 1 else "Negative")