{
 "cells": [
  {
   "cell_type": "markdown",
   "id": "580d78a5",
   "metadata": {},
   "source": [
    "# NLP Classification Assignment\n",
    "SMS Spam Classification"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "89381f89",
   "metadata": {},
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "import nltk,string\n",
    "from nltk.corpus import stopwords\n",
    "from nltk.stem import PorterStemmer\n",
    "from sklearn.model_selection import train_test_split\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer\n",
    "from sklearn.naive_bayes import MultinomialNB\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.svm import LinearSVC\n",
    "from sklearn.metrics import accuracy_score,classification_report,confusion_matrix\n",
    "import matplotlib.pyplot as plt\n",
    "import seaborn as sns\n",
    "nltk.download('stopwords')"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "45e5e3f7",
   "metadata": {},
   "source": [
    "## Load Dataset (spam.csv)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "fa879059",
   "metadata": {},
   "outputs": [],
   "source": [
    "df=pd.read_csv('spam.csv',encoding='latin-1')\n",
    "df=df[['v1','v2']]\n",
    "df.columns=['label','text']\n",
    "df.head()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "9957e864",
   "metadata": {},
   "source": [
    "## Preprocessing"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "8605f3a5",
   "metadata": {},
   "outputs": [],
   "source": [
    "stemmer=PorterStemmer()\n",
    "sw=set(stopwords.words('english'))\n",
    "def preprocess(text):\n",
    "    text=text.lower()\n",
    "    text=''.join(c for c in text if c not in string.punctuation)\n",
    "    words=[stemmer.stem(w) for w in text.split() if w not in sw]\n",
    "    return ' '.join(words)\n",
    "df['clean_text']=df['text'].apply(preprocess)\n",
    "df['label']=df['label'].map({'ham':0,'spam':1})"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "1f6198d9",
   "metadata": {},
   "source": [
    "## Split and TF-IDF"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "17f9a808",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train,X_test,y_train,y_test=train_test_split(df['clean_text'],df['label'],test_size=0.2,random_state=42)\n",
    "tfidf=TfidfVectorizer()\n",
    "X_train=tfidf.fit_transform(X_train)\n",
    "X_test=tfidf.transform(X_test)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "6f32f790",
   "metadata": {},
   "source": [
    "## Train & Evaluate"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "4e39573d",
   "metadata": {},
   "outputs": [],
   "source": [
    "models={\n",
    "'Naive Bayes':MultinomialNB(),\n",
    "'Logistic Regression':LogisticRegression(max_iter=1000),\n",
    "'SVM':LinearSVC()\n",
    "}\n",
    "for name,m in models.items():\n",
    "    m.fit(X_train,y_train)\n",
    "    pred=m.predict(X_test)\n",
    "    print(name,accuracy_score(y_test,pred))\n",
    "    print(classification_report(y_test,pred))\n",
    "cm=confusion_matrix(y_test,pred)\n",
    "sns.heatmap(cm,annot=True,fmt='d')\n",
    "plt.show()"
   ]
  }
 ],
 "metadata": {},
 "nbformat": 4,
 "nbformat_minor": 5
}
