# -*- coding: utf-8 -*-
"""Untitled4.ipynb

Automatically generated by Colab.

Original file is located at
    https://colab.research.google.com/drive/1pUwJfV_ugm1GOPH6_im72td3iL6FQQKs
"""

import re
import json
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.linear_model import LogisticRegression
from sklearn.svm import LinearSVC
from sklearn.pipeline import Pipeline
from sklearn.metrics import accuracy_score, classification_report, precision_recall_fscore_support, confusion_matrix
import joblib
# ===== Upload Dataset File Manually (for Colab / Jupyter) =====
from google.colab import files
uploaded = files.upload()  # سيُطلب منك اختيار الملف من جهازك

# تحديد اسم الملف الذي اخترته
DATA_PATH = list(uploaded.keys())[0]   # سيأخذ أول ملف تقوم برفعه

print(f"Dataset uploaded as: {DATA_PATH}")

# ======== Config ========
DATA_PATH   = "unbalanceddataset.csv"   # path to CSV file
TEXT_COL    = "text"
LABEL_COL   = "sentiment"
TEST_SIZE   = 0.2
RANDOM_STATE = 42
# =========================

# ======== Preprocessing ========
# (Supports Arabic + English)

AR_DIACRITICS = re.compile(r"[\u0617-\u061A\u064B-\u0652]")
AR_TATWEEL    = re.compile(r"\u0640")
URL_RE        = re.compile(r"http\S+|www\.\S+")
TAG_RE        = re.compile(r"[@#]\w+")
MULTI         = re.compile(r"\s+")

def normalize_arabic(s: str) -> str:
    s = AR_TATWEEL.sub('', s)
    s = AR_DIACRITICS.sub('', s)
    s = (s.replace('أ','ا').replace('إ','ا').replace('آ','ا')
            .replace('ى','ي').replace('ؤ','و').replace('ئ','ي')
            .replace('ة','ه'))
    return s

def clean_text(x: str) -> str:
    if not isinstance(x, str):
        x = '' if x is None else str(x)
    x = x.lower()
    x = normalize_arabic(x)
    x = URL_RE.sub(" ", x)
    x = TAG_RE.sub(" ", x)
    x = re.sub(r"[^a-zA-Z\u0600-\u06FF\s]+", " ", x)
    x = MULTI.sub(" ", x).strip()
    return x
# ================================

# ======== 1) Load Dataset ========
df = pd.read_csv(DATA_PATH, encoding="utf-8")
df = df[[TEXT_COL, LABEL_COL]].dropna()
df["clean"] = df[TEXT_COL].apply(clean_text)
print("Loaded:", df.shape)

# ======== 2) Split Train/Test ========
X_train, X_test, y_train, y_test = train_test_split(
    df["clean"], df[LABEL_COL],
    test_size=TEST_SIZE,
    stratify=df[LABEL_COL],
    random_state=RANDOM_STATE
)

# ======== 3) TF-IDF feature extractor ========
tfidf = TfidfVectorizer(ngram_range=(1,2), min_df=2, max_df=0.9)

# ======== 4) Train & Evaluate Models ========
models = {
    "logreg_balanced": LogisticRegression(max_iter=2000, class_weight="balanced"),
    "linear_svc_balanced": LinearSVC(class_weight="balanced")
}

results = {}

for name, clf in models.items():
    pipe = Pipeline([("tfidf", tfidf), ("clf", clf)])
    pipe.fit(X_train, y_train)
    y_pred = pipe.predict(X_test)

    # evaluation
    acc = accuracy_score(y_test, y_pred)
    p, r, f1, _ = precision_recall_fscore_support(y_test, y_pred, average="macro", zero_division=0)
    report = classification_report(y_test, y_pred)
    cm = confusion_matrix(y_test, y_pred).tolist()

    results[name] = {
        "accuracy": acc,
        "precision_macro": p,
        "recall_macro": r,
        "f1_macro": f1,
        "confusion_matrix": cm,
        "report": report
    }

    # save model
    joblib.dump(pipe, f"{name}.joblib")
    print(f"\n=== {name} ===")
    print(f"Accuracy: {acc:.4f} | F1-macro: {f1:.4f}")
    print(report)

# ======== 5) Save evaluation summary ========
with open("eval_summary.json", "w", encoding="utf-8") as f:
    json.dump(results, f, ensure_ascii=False, indent=2)

print("\nArtifacts saved:")
print(" - logreg_balanced.joblib")
print(" - linear_svc_balanced.joblib")
print(" - eval_summary.json")