{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "cba3bba5",
   "metadata": {},
   "outputs": [
    {
     "name": "stderr",
     "output_type": "stream",
     "text": [
      "[nltk_data] Downloading package punkt to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package punkt is already up-to-date!\n",
      "[nltk_data] Downloading package punkt_tab to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package punkt_tab is already up-to-date!\n",
      "[nltk_data] Downloading package stopwords to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package stopwords is already up-to-date!\n",
      "[nltk_data] Downloading package wordnet to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package wordnet is already up-to-date!\n",
      "[nltk_data] Downloading package punkt to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package punkt is already up-to-date!\n",
      "[nltk_data] Downloading package stopwords to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package stopwords is already up-to-date!\n",
      "[nltk_data] Downloading package wordnet to\n",
      "[nltk_data]     C:\\Users\\mmoghyiri\\AppData\\Roaming\\nltk_data...\n",
      "[nltk_data]   Package wordnet is already up-to-date!\n"
     ]
    },
    {
     "data": {
      "text/plain": [
       "True"
      ]
     },
     "execution_count": 1,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "\n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "import re\n",
    "import nltk\n",
    "import seaborn as sns\n",
    "import matplotlib.pyplot as plt\n",
    "\n",
    "from sklearn.model_selection import train_test_split, GridSearchCV\n",
    "from sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\n",
    "from sklearn.linear_model import LogisticRegression\n",
    "from sklearn.naive_bayes import MultinomialNB\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report, confusion_matrix\n",
    "\n",
    "from gensim.models import Word2Vec\n",
    "from nltk.tokenize import word_tokenize\n",
    "from nltk.corpus import stopwords\n",
    "from nltk.stem import PorterStemmer, WordNetLemmatizer\n",
    "import nltk\n",
    "for pkg in [\"punkt\", \"punkt_tab\", \"stopwords\", \"wordnet\"]:\n",
    "    nltk.download(pkg)\n",
    "\n",
    "from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report, confusion_matrix\n",
    "import seaborn as sns\n",
    "import matplotlib.pyplot as plt\n",
    "    \n",
    "nltk.download('punkt')\n",
    "nltk.download('stopwords')\n",
    "nltk.download('wordnet')\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "e65118fd",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "   label                                            message  label_num\n",
      "0    ham  Go until jurong point, crazy.. Available only ...          0\n",
      "1    ham                      Ok lar... Joking wif u oni...          0\n",
      "2   spam  Free entry in 2 a wkly comp to win FA Cup fina...          1\n",
      "3    ham  U dun say so early hor... U c already then say...          0\n",
      "4    ham  Nah I don't think he goes to usf, he lives aro...          0\n",
      "5   spam  FreeMsg Hey there darling it's been 3 week's n...          1\n",
      "6    ham  Even my brother is not like to speak with me. ...          0\n",
      "7    ham  As per your request 'Melle Melle (Oru Minnamin...          0\n",
      "8   spam  WINNER!! As a valued network customer you have...          1\n",
      "9   spam  Had your mobile 11 months or more? U R entitle...          1\n",
      "10   ham  I'm gonna be home soon and i don't want to tal...          0\n",
      "11  spam  SIX chances to win CASH! From 100 to 20,000 po...          1\n",
      "12  spam  URGENT! You have won a 1 week FREE membership ...          1\n",
      "13   ham  I've been searching for the right words to tha...          0\n",
      "14   ham                I HAVE A DATE ON SUNDAY WITH WILL!!          0\n",
      "15  spam  XXXMobileMovieClub: To use your credit, click ...          1\n",
      "16   ham                         Oh k...i'm watching here:)          0\n",
      "17   ham  Eh u remember how 2 spell his name... Yes i di...          0\n",
      "18   ham  Fine if thatÃ¥Ãs the way u feel. ThatÃ¥Ãs th...          0\n",
      "19  spam  England v Macedonia - dont miss the goals/team...          1\n",
      "\n",
      "label\n",
      "ham     4825\n",
      "spam     747\n",
      "Name: count, dtype: int64\n",
      "\n",
      "label_num\n",
      "0    4825\n",
      "1     747\n",
      "Name: count, dtype: int64\n",
      "\n"
     ]
    }
   ],
   "source": [
    "\n",
    "# load data\n",
    "\n",
    "df = pd.read_csv(\"spam_sms.csv\", encoding=\"latin-1\")\n",
    "df.columns = [\"label\", \"message\"]\n",
    "df['label_num'] = df['label'].map({'ham': 0, 'spam': 1})\n",
    "\n",
    "print(df.head(20), end=\"\\n\\n\")\n",
    "print(df['label'].value_counts(), end=\"\\n\\n\")\n",
    "print(df['label_num'].value_counts(), end=\"\\n\\n\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "1068e1c8-cc12-4388-a8f0-858f70d2bec6",
   "metadata": {},
   "outputs": [],
   "source": []
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "f0aafdff-0420-44d3-9536-5fa4aa368d94",
   "metadata": {},
   "outputs": [],
   "source": []
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "09913d2f",
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "    label_num  label_num                                         clean_text\n",
      "0           0          0  go jurong point crazy available bugis n great ...\n",
      "1           0          0                            ok lar joking wif u oni\n",
      "2           1          1  free entry wkly comp win fa cup final tkts st ...\n",
      "3           0          0                u dun say early hor u c already say\n",
      "4           0          0           nah dont think go usf life around though\n",
      "5           1          1  freemsg hey darling week word back id like fun...\n",
      "6           0          0      even brother like speak treat like aid patent\n",
      "7           0          0  per request melle melle oru minnaminunginte nu...\n",
      "8           1          1  winner valued network customer selected receiv...\n",
      "9           1          1  mobile month u r entitled update latest colour...\n",
      "10          0          0  im gon na home soon dont want talk stuff anymo...\n",
      "11          1          1  six chance win cash pound txt csh send cost pd...\n",
      "12          1          1  urgent week free membership prize jackpot txt ...\n",
      "13          0          0  ive searching right word thank breather promis...\n",
      "14          0          0                                        date sunday\n",
      "15          1          1  xxxmobilemovieclub use credit click wap link n...\n",
      "16          0          0                                    oh kim watching\n",
      "17          0          0  eh u remember spell name yes v naughty make v wet\n",
      "18          0          0             fine thats way u feel thats way gota b\n",
      "19          1          1  england v macedonia dont miss goalsteam news t...\n",
      "20          0          0                               seriously spell name\n",
      "21          0          0                    im going try month ha ha joking\n",
      "22          0          0                       pay first lar da stock comin\n",
      "23          0          0  aft finish lunch go str lor ard smth lor u fin...\n",
      "24          0          0                 ffffffffff alright way meet sooner\n",
      "25          0          0  forced eat slice im really hungry tho suck mar...\n",
      "26          0          0                              lol always convincing\n",
      "27          0          0  catch bus frying egg make tea eating mom left ...\n",
      "28          0          0    im back amp packing car ill let know there room\n",
      "29          0          0           ahhh work vaguely remember feel like lol\n",
      "30          0          0  wait thats still clear sure sarcastic thats x ...\n",
      "31          0          0  yeah got v apologetic n fallen actin like spoi...\n",
      "32          0          0                                    k tell anything\n",
      "33          0          0                fear fainting housework quick cuppa\n",
      "34          1          1  thanks subscription ringtone uk mobile charged...\n",
      "35          0          0  yup ok go home look timing msg xuhui going lea...\n",
      "36          0          0                    oops ill let know roommate done\n",
      "37          0          0                                   see letter b car\n",
      "38          0          0                              anything lor u decide\n",
      "39          0          0  hello hows saturday go texting see youd decide...\n",
      "40          0          0  pls go ahead watt wanted sure great weekend ab...\n",
      "41          0          0  forget tell want need crave love sweet arabian...\n",
      "42          1          1  rodger burn msg tried call reply sm free nokia...\n",
      "43          0          0                                             seeing\n",
      "44          0          0         great hope like man well endowed ltgt inch\n",
      "45          0          0                           callsmessagesmissed call\n",
      "46          0          0               didnt get hep b immunisation nigeria\n",
      "47          0          0                         fair enough anything going\n",
      "48          0          0  yeah hopefully tyler cant could maybe ask arou...\n",
      "49          0          0  u dont know stubborn didnt even want go hospit...\n"
     ]
    }
   ],
   "source": [
    "# preprocessing\n",
    "\n",
    "stop_words = set(stopwords.words('english'))\n",
    "stemmer = PorterStemmer()\n",
    "lemmatizer = WordNetLemmatizer()\n",
    "\n",
    "def preprocess_text(text):\n",
    "    text = text.lower()\n",
    "    text = re.sub(r'http\\S+|www\\.\\S+', '', text)\n",
    "    text = re.sub(r'<.*?>', '', text)\n",
    "    text = re.sub(r'[^a-zA-Z\\s]', '', text)\n",
    "    text = str(text).lower().strip()\n",
    "    text = url_pat.sub('', text)\n",
    "    text = html_pat.sub('', text)\n",
    "    text = non_alpha_pat.sub(' ', text)\n",
    "\n",
    "    tokens = word_tokenize(text, preserve_line=True)\n",
    "    tokens = [t for t in tokens if t not in stop_words]\n",
    "    tokens = [lemmatizer.lemmatize(t) for t in tokens]\n",
    "    return ' '.join(tokens)\n",
    "\n",
    "df['clean_text'] = df['message'].apply(preprocess_text)\n",
    "print(df[['label_num','label_num', 'clean_text']].head(50))\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "aa6b8777",
   "metadata": {},
   "outputs": [],
   "source": [
    "# train & split\n",
    "\n",
    "X = df['clean_text']\n",
    "y = df['label_num']\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n",
    "\n",
    "len(X_train), len(X_test), np.unique(y, return_counts=True)\n",
    "\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "ecb0ed18",
   "metadata": {},
   "outputs": [],
   "source": [
    "vectorizer = TfidfVectorizer(ngram_range=(1,2))\n",
    "X_train_vec = vectorizer.fit_transform(X_train)\n",
    "X_test_vec = vectorizer.transform(X_test)\n",
    "\n",
    "print(\"TF-IDF shape:\", X_train_vec.shape)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "87c61a2e",
   "metadata": {},
   "outputs": [],
   "source": [
    "# evaluate\n",
    "\n",
    "def evaluate_model(model, X_test, y_test, name):\n",
    "    y_pred = model.predict(X_test)\n",
    "    print(f\"\\n===== {name} =====\")\n",
    "    print(\"Accuracy:\", accuracy_score(y_test, y_pred))\n",
    "    print(\"Precision:\", precision_score(y_test, y_pred, zero_division=0))\n",
    "    print(\"Recall:\", recall_score(y_test, y_pred, zero_division=0))\n",
    "    print(\"F1:\", f1_score(y_test, y_pred, zero_division=0))\n",
    "    print(\"\\nClassification Report:\\n\", classification_report(y_test, y_pred, zero_division=0))\n",
    "    cm = confusion_matrix(y_test, y_pred)\n",
    "    sns.heatmap(cm, annot=True, fmt=\"d\", cbar=False)\n",
    "    plt.title(f\"Confusion Matrix - {name}\")\n",
    "    plt.xlabel(\"Predicted\")\n",
    "    plt.ylabel(\"Actual\")\n",
    "    plt.show()\n",
    "\n",
    "lr = LogisticRegression(max_iter=200)\n",
    "lr.fit(X_train_vec, y_train)\n",
    "evaluate_model(lr, X_test_vec, y_test, \"Logistic Regression\")\n",
    "\n",
    "nb = MultinomialNB()\n",
    "nb.fit(X_train_vec, y_train)\n",
    "evaluate_model(nb, X_test_vec, y_test, \"Naive Bayes\")\n",
    "\n",
    "rf = RandomForestClassifier(n_estimators=200, max_depth=20, random_state=42)\n",
    "rf.fit(X_train_vec, y_train)\n",
    "evaluate_model(rf, X_test_vec, y_test, \"Random Forest\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "98cf45d7",
   "metadata": {},
   "outputs": [],
   "source": [
    "df['tokens'] = df['clean_text'].apply(lambda x: word_tokenize(x))\n",
    "\n",
    "w2v_model = Word2Vec(sentences=df['tokens'], vector_size=100, window=5, min_count=1, sg=0)\n",
    "\n",
    "def get_sentence_vector(tokens, model):\n",
    "    vectors = [model.wv[w] for w in tokens if w in model.wv]\n",
    "    return np.mean(vectors, axis=0) if vectors else np.zeros(model.vector_size)\n",
    "\n",
    "df['vector'] = df['tokens'].apply(lambda x: get_sentence_vector(x, w2v_model))\n",
    "\n",
    "X = np.vstack(df['vector'].values)\n",
    "y = df['label_num'].values\n",
    "\n",
    "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n",
    "\n",
    "lr_w2v = LogisticRegression(max_iter=200)\n",
    "lr_w2v.fit(X_train, y_train)\n",
    "evaluate_model(lr_w2v, X_test, y_test, \"Logistic Regression (Word2Vec)\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "f168ba8a-681f-4a47-a9f9-da37dcd6fe63",
   "metadata": {},
   "outputs": [],
   "source": []
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "6878fff8-d5fe-4863-9a96-31a6740302f4",
   "metadata": {},
   "outputs": [],
   "source": []
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3 (ipykernel)",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.11.13"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
