import pandas as pd
import re

# Load dataset
df = pd.read_excel("unbalanceddataset.xlsx")

# Rename columns if needed
df = df[['text', 'sentiment']]

# Convert text to string
df['text'] = df['text'].astype(str)

# Text preprocessing function
def preprocess_text(text):
    text = text.lower()                              # lowercase
    text = re.sub(r"http\S+|www\S+", "", text)       # remove URLs
    text = re.sub(r"@\w+", "", text)                 # remove mentions
    text = re.sub(r"#", "", text)                    # remove hashtag symbol
    text = re.sub(r"[^a-z\s']", "", text)            # keep letters only
    text = re.sub(r"\s+", " ", text).strip()         # remove extra spaces
    return text

# Apply preprocessing
df['clean_text'] = df['text'].apply(preprocess_text)

# Check result
print(df.head())
from sklearn.model_selection import train_test_split

# Features and labels
X = df['clean_text']      # preprocessed text
y = df['sentiment']       # labels

# Step 1: Train + Temp (85%)
X_train, X_temp, y_train, y_temp = train_test_split(
    X,
    y,
    test_size=0.30,
    random_state=42,
    stratify=y
)

# Step 2: Validation (15%) and Test (15%)
X_val, X_test, y_val, y_test = train_test_split(
    X_temp,
    y_temp,
    test_size=0.50,
    random_state=42,
    stratify=y_temp
)

# Check sizes
print("Training set:", X_train.shape)
print("Validation set:", X_val.shape)
print("Test set:", X_test.shape)


import os, textwrap

code = r'''"""
NLP Text Classification: Feature Representation + Model Training/Evaluation
=========================================================================
Works with your Excel dataset (e.g., unbalanceddataset.xlsx) that contains:
- text column:    "text"
- label column:   "sentiment"

Includes (as requested):
1) Feature representation methods:
   - Bag of Words (CountVectorizer) word n-grams
   - TF-IDF word n-grams
   - TF-IDF character n-grams
2) Train & evaluate multiple models:
   - Logistic Regression (class_weight=balanced)
   - Linear SVM (LinearSVC, class_weight=balanced)
   - Multinomial Naive Bayes
3) Reports metrics:
   - Accuracy, weighted F1, macro F1
   - classification_report
   - confusion_matrix
4) Saves the best-performing pipeline (by validation weighted F1) to disk.

Run:
    python nlp_train_eval.py --data unbalanceddataset.xlsx

Dependencies:
    pip install pandas numpy scikit-learn openpyxl joblib
"""

from __future__ import annotations

import argparse
import re
from dataclasses import dataclass
from typing import Dict, List, Tuple, Optional

import pandas as pd

from sklearn.model_selection import train_test_split
from sklearn.pipeline import Pipeline
from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer
from sklearn.metrics import accuracy_score, f1_score, classification_report, confusion_matrix
from sklearn.linear_model import LogisticRegression
from sklearn.svm import LinearSVC
from sklearn.naive_bayes import MultinomialNB
import joblib


# ----------------------------
# 1) Load + Preprocess
# ----------------------------

_URL_RE = re.compile(r"http\\S+|www\\.\\S+")
_MENTION_RE = re.compile(r"@\\w+")
_HASHTAG_RE = re.compile(r"#(\\w+)")

def preprocess_text(text: str) -> str:
    """
    Basic preprocessing:
    - lowercase
    - remove URLs and @mentions
    - keep hashtag words but remove '#'
    - remove non-letter chars (keeps apostrophes)
    - normalize whitespace
    """
    s = str(text).lower()
    s = _URL_RE.sub(" ", s)
    s = _MENTION_RE.sub(" ", s)
    s = _HASHTAG_RE.sub(r"\\1", s)
    s = re.sub(r"[^a-z\\s']", " ", s)
    s = re.sub(r"\\s+", " ", s).strip()
    return s


def load_excel(path: str, text_col: str, label_col: str) -> Tuple[pd.Series, pd.Series]:
    df = pd.read_excel(path)
    if text_col not in df.columns or label_col not in df.columns:
        raise ValueError(f"Expected columns '{text_col}' and '{label_col}'. Found: {list(df.columns)}")
    X = df[text_col].astype(str).map(preprocess_text)
    y = df[label_col].astype(str)
    return X, y


# ----------------------------
# 2) Split
# ----------------------------

@dataclass
class Splits:
    X_train: pd.Series
    X_val: pd.Series
    X_test: pd.Series
    y_train: pd.Series
    y_val: pd.Series
    y_test: pd.Series


def stratified_split(X: pd.Series, y: pd.Series, seed: int = 42,
                     test_size: float = 0.15, val_size: float = 0.15) -> Splits:
    if test_size + val_size >= 1.0:
        raise ValueError("test_size + val_size must be < 1.0")

    X_train, X_tmp, y_train, y_tmp = train_test_split(
        X, y, test_size=(test_size + val_size), random_state=seed, stratify=y
    )
    rel_test_size = test_size / (test_size + val_size)
    X_val, X_test, y_val, y_test = train_test_split(
        X_tmp, y_tmp, test_size=rel_test_size, random_state=seed, stratify=y_tmp
    )
    return Splits(X_train, X_val, X_test, y_train, y_val, y_test)


# ----------------------------
# 3) Feature representations + models
# ----------------------------

def candidate_pipelines() -> List[Tuple[str, Pipeline]]:
    """Build combinations of feature representation methods + classifiers."""
    vectorizers: Dict[str, object] = {
        "bow_word_1_2": CountVectorizer(ngram_range=(1, 2), min_df=2, max_df=0.95),
        "tfidf_word_1_2": TfidfVectorizer(ngram_range=(1, 2), min_df=2, max_df=0.95),
        "tfidf_char_3_5": TfidfVectorizer(analyzer="char", ngram_range=(3, 5), min_df=2, max_df=0.95),
    }

    models: Dict[str, object] = {
        "logreg_bal": LogisticRegression(max_iter=3000, class_weight="balanced"),
        "linearsvc_bal": LinearSVC(class_weight="balanced", dual="auto"),
        "mnb": MultinomialNB(),
    }

    cands: List[Tuple[str, Pipeline]] = []
    for v_name, vec in vectorizers.items():
        for m_name, model in models.items():
            name = f"{v_name}+{m_name}"
            pipe = Pipeline([("vec", vec), ("clf", model)])
            cands.append((name, pipe))
    return cands


def score_model(pipe: Pipeline, X: pd.Series, y: pd.Series) -> Dict[str, float]:
    pred = pipe.predict(X)
    return {
        "accuracy": float(accuracy_score(y, pred)),
        "f1_weighted": float(f1_score(y, pred, average="weighted")),
        "f1_macro": float(f1_score(y, pred, average="macro")),
    }


# ----------------------------
# 4) Train + Evaluate
# ----------------------------

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--data", type=str, default="unbalanceddataset.xlsx")
    ap.add_argument("--text_col", type=str, default="text")
    ap.add_argument("--label_col", type=str, default="sentiment")
    ap.add_argument("--seed", type=int, default=42)
    ap.add_argument("--save_model", type=str, default="best_nlp_model.joblib")
    args = ap.parse_args()

    # Load + preprocess
    X, y = load_excel(args.data, args.text_col, args.label_col)
    print(f"Loaded: {args.data} | rows={len(X)}")
    print("\\nLabel distribution:")
    print(y.value_counts(dropna=False))

    # Split
    splits = stratified_split(X, y, seed=args.seed, test_size=0.15, val_size=0.15)
    print(f"\\nSplit sizes: train={len(splits.X_train)}, val={len(splits.X_val)}, test={len(splits.X_test)}")

    # Train candidates and pick best by validation weighted F1
    best_name: Optional[str] = None
    best_pipe: Optional[Pipeline] = None
    best_val_f1: float = -1.0

    print("\\nTraining candidates (validation metrics):")
    for name, pipe in candidate_pipelines():
        pipe.fit(splits.X_train, splits.y_train)
        metrics = score_model(pipe, splits.X_val, splits.y_val)
        print(f"  {name:24s} | acc={metrics['accuracy']:.4f} | f1_w={metrics['f1_weighted']:.4f} | f1_m={metrics['f1_macro']:.4f}")

        if metrics["f1_weighted"] > best_val_f1:
            best_val_f1 = metrics["f1_weighted"]
            best_name = name
            best_pipe = pipe

    assert best_pipe is not None and best_name is not None
    print(f"\\nBest model on validation: {best_name} (weighted F1={best_val_f1:.4f})")

    # Evaluate on test
    test_pred = best_pipe.predict(splits.X_test)
    print("\\n=== TEST EVALUATION ===")
    print(f"Accuracy: {accuracy_score(splits.y_test, test_pred):.4f}")
    print(f"Weighted F1: {f1_score(splits.y_test, test_pred, average='weighted'):.4f}")
    print(f"Macro F1: {f1_score(splits.y_test, test_pred, average='macro'):.4f}\\n")
    print(classification_report(splits.y_test, test_pred, digits=4))

    labels = sorted(pd.unique(y))
    cm = confusion_matrix(splits.y_test, test_pred, labels=labels)
    print("Confusion Matrix (rows=true, cols=pred):")
    print("labels:", labels)
    print(cm)

    # Save best pipeline
    joblib.dump(best_pipe, args.save_model)
    print(f"\\nSaved best pipeline to: {args.save_model}")


if __name__ == "__main__":
    main()
'''
import pandas as pd
import re
from sklearn.model_selection import train_test_split
from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer
from sklearn.pipeline import Pipeline
from sklearn.linear_model import LogisticRegression
from sklearn.svm import LinearSVC
from sklearn.naive_bayes import MultinomialNB
from sklearn.metrics import accuracy_score, f1_score, classification_report, confusion_matrix

