{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14169237,"sourceType":"datasetVersion","datasetId":9031727}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_recall_fscore_support,\n    classification_report,\n    confusion_matrix,\n    roc_auc_score\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:17:36.593910Z","iopub.execute_input":"2025-12-15T22:17:36.594239Z","iopub.status.idle":"2025-12-15T22:17:41.603834Z","shell.execute_reply.started":"2025-12-15T22:17:36.594207Z","shell.execute_reply":"2025-12-15T22:17:41.602700Z"}},"outputs":[],"execution_count":1},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/project2-dataset\"\n\nprint(\"Files in dataset folder:\")\nprint(os.listdir(DATA_DIR))\n\n# If the filename differs, pick it from the printed list\nDATA_PATH = os.path.join(DATA_DIR, \"unbalanceddataset (1).csv\")\n\ndf = pd.read_csv(DATA_PATH)\n\nprint(\"Shape:\", df.shape)\nprint(\"Columns:\", df.columns.tolist())\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:17:41.608579Z","iopub.execute_input":"2025-12-15T22:17:41.608856Z","iopub.status.idle":"2025-12-15T22:17:41.675514Z","shell.execute_reply.started":"2025-12-15T22:17:41.608833Z","shell.execute_reply":"2025-12-15T22:17:41.674142Z"}},"outputs":[{"name":"stdout","text":"Files in dataset folder:\n['unbalanceddataset (1).csv']\nShape: (2600, 2)\nColumns: ['text', 'sentiment']\n","output_type":"stream"},{"execution_count":2,"output_type":"execute_result","data":{"text/plain":"                                                text sentiment\n0  Java Concurrency in Practice is probably the b...  positive\n1    haha aww hun i bet you are more creative tha...  positive\n2  _pickle lol, thank you very much Hope you`re h...  positive\n3  Out for an evening on the town with jeremy. Sa...  negative\n4   - just took over the #1 Most Endorsed spot on...  positive","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>text</th>\n      <th>sentiment</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>Java Concurrency in Practice is probably the b...</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>haha aww hun i bet you are more creative tha...</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>_pickle lol, thank you very much Hope you`re h...</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>Out for an evening on the town with jeremy. Sa...</td>\n      <td>negative</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>- just took over the #1 Most Endorsed spot on...</td>\n      <td>positive</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":2},{"cell_type":"markdown","source":"# Quick check + label encoding","metadata":{}},{"cell_type":"code","source":"# Keep only required columns and drop missing values\ndf = df.dropna(subset=[\"text\", \"sentiment\"]).copy()\n\nprint(\"Class distribution:\")\nprint(df[\"sentiment\"].value_counts())\n\n# Encode labels\nlabel_map = {\"negative\": 0, \"positive\": 1}\ndf[\"y\"] = df[\"sentiment\"].map(label_map)\n\n# Safety check\nassert df[\"y\"].isna().sum() == 0, \"Unmapped labels found. Check df['sentiment'].unique().\"\nprint(\"Unique labels:\", df[\"sentiment\"].unique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:18:09.696937Z","iopub.execute_input":"2025-12-15T22:18:09.697264Z","iopub.status.idle":"2025-12-15T22:18:09.722572Z","shell.execute_reply.started":"2025-12-15T22:18:09.697240Z","shell.execute_reply":"2025-12-15T22:18:09.721513Z"}},"outputs":[{"name":"stdout","text":"Class distribution:\nsentiment\npositive    2000\nnegative     600\nName: count, dtype: int64\nUnique labels: ['positive' 'negative']\n","output_type":"stream"}],"execution_count":3},{"cell_type":"markdown","source":"# Apply data preprocessing","metadata":{}},{"cell_type":"code","source":"def clean_text(text: str) -> str:\n    text = str(text).lower()\n    text = re.sub(r\"http\\S+|www\\S+\", \" \", text)     # URLs\n    text = re.sub(r\"<.*?>\", \" \", text)             # HTML tags\n    text = re.sub(r\"@\\w+|#\\w+\", \" \", text)         # mentions/hashtags\n    text = re.sub(r\"[^a-z\\s]\", \" \", text)          # keep letters + spaces\n    text = re.sub(r\"\\s+\", \" \", text).strip()       # normalize spaces\n    return text\n\ndf[\"text_clean\"] = df[\"text\"].apply(clean_text)\n\ndf[[\"text\", \"text_clean\", \"sentiment\"]].head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:18:36.216359Z","iopub.execute_input":"2025-12-15T22:18:36.216766Z","iopub.status.idle":"2025-12-15T22:18:36.272411Z","shell.execute_reply.started":"2025-12-15T22:18:36.216736Z","shell.execute_reply":"2025-12-15T22:18:36.271038Z"}},"outputs":[{"execution_count":4,"output_type":"execute_result","data":{"text/plain":"                                                text  \\\n0  Java Concurrency in Practice is probably the b...   \n1    haha aww hun i bet you are more creative tha...   \n2  _pickle lol, thank you very much Hope you`re h...   \n3  Out for an evening on the town with jeremy. Sa...   \n4   - just took over the #1 Most Endorsed spot on...   \n\n                                          text_clean sentiment  \n0  java concurrency in practice is probably the b...  positive  \n1   haha aww hun i bet you are more creative than me  positive  \n2  pickle lol thank you very much hope you re hav...  positive  \n3  out for an evening on the town with jeremy sad...  negative  \n4  just took over the most endorsed spot on twind...  positive  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>text</th>\n      <th>text_clean</th>\n      <th>sentiment</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>Java Concurrency in Practice is probably the b...</td>\n      <td>java concurrency in practice is probably the b...</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>haha aww hun i bet you are more creative tha...</td>\n      <td>haha aww hun i bet you are more creative than me</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>_pickle lol, thank you very much Hope you`re h...</td>\n      <td>pickle lol thank you very much hope you re hav...</td>\n      <td>positive</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>Out for an evening on the town with jeremy. Sa...</td>\n      <td>out for an evening on the town with jeremy sad...</td>\n      <td>negative</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>- just took over the #1 Most Endorsed spot on...</td>\n      <td>just took over the most endorsed spot on twind...</td>\n      <td>positive</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":4},{"cell_type":"markdown","source":"# Split the dataset","metadata":{}},{"cell_type":"code","source":"X = df[\"text_clean\"].values\ny = df[\"y\"].values\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y,\n    test_size=0.20,\n    random_state=42,\n    stratify=y\n)\n\nprint(\"Train size:\", len(X_train), \"Test size:\", len(X_test))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:19:04.951172Z","iopub.execute_input":"2025-12-15T22:19:04.951679Z","iopub.status.idle":"2025-12-15T22:19:04.963174Z","shell.execute_reply.started":"2025-12-15T22:19:04.951644Z","shell.execute_reply":"2025-12-15T22:19:04.961895Z"}},"outputs":[{"name":"stdout","text":"Train size: 2080 Test size: 520\n","output_type":"stream"}],"execution_count":5},{"cell_type":"markdown","source":"# Feature representation methods (TF-IDF)","metadata":{}},{"cell_type":"code","source":"# Word-level TF-IDF\ntfidf_word = TfidfVectorizer(ngram_range=(1,2), max_features=5000)\nX_train_word = tfidf_word.fit_transform(X_train)\nX_test_word  = tfidf_word.transform(X_test)\n\n# Character-level TF-IDF\ntfidf_char = TfidfVectorizer(analyzer=\"char\", ngram_range=(3,5), max_features=8000)\nX_train_char = tfidf_char.fit_transform(X_train)\nX_test_char  = tfidf_char.transform(X_test)\n\nprint(\"Word TF-IDF shape:\", X_train_word.shape)\nprint(\"Char TF-IDF shape:\", X_train_char.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:19:34.153089Z","iopub.execute_input":"2025-12-15T22:19:34.153497Z","iopub.status.idle":"2025-12-15T22:19:34.825602Z","shell.execute_reply.started":"2025-12-15T22:19:34.153451Z","shell.execute_reply":"2025-12-15T22:19:34.824111Z"}},"outputs":[{"name":"stdout","text":"Word TF-IDF shape: (2080, 5000)\nChar TF-IDF shape: (2080, 8000)\n","output_type":"stream"}],"execution_count":6},{"cell_type":"markdown","source":"# Train and evaluate models","metadata":{}},{"cell_type":"code","source":"def evaluate_model(name, model, Xtr, Xte, ytr, yte):\n    model.fit(Xtr, ytr)\n    pred = model.predict(Xte)\n\n    acc = accuracy_score(yte, pred)\n    p, r, f1, _ = precision_recall_fscore_support(yte, pred, average=\"binary\", zero_division=0)\n\n    auc = None\n    if hasattr(model, \"predict_proba\"):\n        proba = model.predict_proba(Xte)[:, 1]\n        auc = roc_auc_score(yte, proba)\n    elif hasattr(model, \"decision_function\"):\n        scores = model.decision_function(Xte)\n        auc = roc_auc_score(yte, scores)\n\n    print(\"=\"*80)\n    print(name)\n    if auc is not None:\n        print(f\"Accuracy: {acc:.4f} | Precision: {p:.4f} | Recall: {r:.4f} | F1: {f1:.4f} | ROC-AUC: {auc:.4f}\")\n    else:\n        print(f\"Accuracy: {acc:.4f} | Precision: {p:.4f} | Recall: {r:.4f} | F1: {f1:.4f}\")\n\n    print(\"\\nClassification Report:\\n\",\n          classification_report(yte, pred, target_names=[\"negative\",\"positive\"], zero_division=0))\n    print(\"Confusion Matrix:\\n\", confusion_matrix(yte, pred))\n    return model, f1, auc if auc is not None else -1\n\n\nresults = []\n\n# Word TF-IDF models\nlr_model, lr_f1, lr_auc = evaluate_model(\n    \"Logistic Regression (Word TF-IDF)\",\n    LogisticRegression(max_iter=2000, class_weight=\"balanced\"),\n    X_train_word, X_test_word, y_train, y_test\n)\nresults.append((\"Logistic Regression (Word TF-IDF)\", lr_f1, lr_auc, lr_model))\n\nsvm_word_model, svm_word_f1, svm_word_auc = evaluate_model(\n    \"Linear SVM (Word TF-IDF)\",\n    LinearSVC(class_weight=\"balanced\"),\n    X_train_word, X_test_word, y_train, y_test\n)\nresults.append((\"Linear SVM (Word TF-IDF)\", svm_word_f1, svm_word_auc, svm_word_model))\n\nnb_model, nb_f1, nb_auc = evaluate_model(\n    \"Multinomial Naive Bayes (Word TF-IDF)\",\n    MultinomialNB(),\n    X_train_word, X_test_word, y_train, y_test\n)\nresults.append((\"Multinomial Naive Bayes (Word TF-IDF)\", nb_f1, nb_auc, nb_model))\n\nrf_model, rf_f1, rf_auc = evaluate_model(\n    \"Random Forest (Word TF-IDF)\",\n    RandomForestClassifier(n_estimators=300, random_state=42, class_weight=\"balanced\"),\n    X_train_word, X_test_word, y_train, y_test\n)\nresults.append((\"Random Forest (Word TF-IDF)\", rf_f1, rf_auc, rf_model))\n\n# Char TF-IDF model\nsvm_char_model, svm_char_f1, svm_char_auc = evaluate_model(\n    \"Linear SVM (Char TF-IDF)\",\n    LinearSVC(class_weight=\"balanced\"),\n    X_train_char, X_test_char, y_train, y_test\n)\nresults.append((\"Linear SVM (Char TF-IDF)\", svm_char_f1, svm_char_auc, svm_char_model))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:20:10.051215Z","iopub.execute_input":"2025-12-15T22:20:10.051580Z","iopub.status.idle":"2025-12-15T22:20:14.734422Z","shell.execute_reply.started":"2025-12-15T22:20:10.051555Z","shell.execute_reply":"2025-12-15T22:20:14.733149Z"}},"outputs":[{"name":"stdout","text":"================================================================================\nLogistic Regression (Word TF-IDF)\nAccuracy: 0.8212 | Precision: 0.9183 | Recall: 0.8425 | F1: 0.8787 | ROC-AUC: 0.8775\n\nClassification Report:\n               precision    recall  f1-score   support\n\n    negative       0.59      0.75      0.66       120\n    positive       0.92      0.84      0.88       400\n\n    accuracy                           0.82       520\n   macro avg       0.75      0.80      0.77       520\nweighted avg       0.84      0.82      0.83       520\n\nConfusion Matrix:\n [[ 90  30]\n [ 63 337]]\n================================================================================\nLinear SVM (Word TF-IDF)\nAccuracy: 0.8308 | Precision: 0.8881 | Recall: 0.8925 | F1: 0.8903 | ROC-AUC: 0.8648\n\nClassification Report:\n               precision    recall  f1-score   support\n\n    negative       0.64      0.62      0.63       120\n    positive       0.89      0.89      0.89       400\n\n    accuracy                           0.83       520\n   macro avg       0.76      0.76      0.76       520\nweighted avg       0.83      0.83      0.83       520\n\nConfusion Matrix:\n [[ 75  45]\n [ 43 357]]\n================================================================================\nMultinomial Naive Bayes (Word TF-IDF)\nAccuracy: 0.7808 | Precision: 0.7782 | Recall: 1.0000 | F1: 0.8753 | ROC-AUC: 0.8804\n\nClassification Report:\n               precision    recall  f1-score   support\n\n    negative       1.00      0.05      0.10       120\n    positive       0.78      1.00      0.88       400\n\n    accuracy                           0.78       520\n   macro avg       0.89      0.53      0.49       520\nweighted avg       0.83      0.78      0.70       520\n\nConfusion Matrix:\n [[  6 114]\n [  0 400]]\n================================================================================\nRandom Forest (Word TF-IDF)\nAccuracy: 0.8250 | Precision: 0.8381 | Recall: 0.9575 | F1: 0.8938 | ROC-AUC: 0.8680\n\nClassification Report:\n               precision    recall  f1-score   support\n\n    negative       0.73      0.38      0.50       120\n    positive       0.84      0.96      0.89       400\n\n    accuracy                           0.82       520\n   macro avg       0.78      0.67      0.70       520\nweighted avg       0.81      0.82      0.80       520\n\nConfusion Matrix:\n [[ 46  74]\n [ 17 383]]\n================================================================================\nLinear SVM (Char TF-IDF)\nAccuracy: 0.8346 | Precision: 0.8925 | Recall: 0.8925 | F1: 0.8925 | ROC-AUC: 0.8793\n\nClassification Report:\n               precision    recall  f1-score   support\n\n    negative       0.64      0.64      0.64       120\n    positive       0.89      0.89      0.89       400\n\n    accuracy                           0.83       520\n   macro avg       0.77      0.77      0.77       520\nweighted avg       0.83      0.83      0.83       520\n\nConfusion Matrix:\n [[ 77  43]\n [ 43 357]]\n","output_type":"stream"}],"execution_count":7},{"cell_type":"markdown","source":"# Save best model + vectorizer","metadata":{}},{"cell_type":"code","source":"import joblib\n\n# Pick best by F1 (requirement-safe)\nbest = sorted(results, key=lambda x: x[1], reverse=True)[0]\nbest_name, best_f1, best_auc, best_model = best\n\nprint(\"Best model:\", best_name, \"| F1:\", best_f1, \"| AUC:\", best_auc)\n\n# Save matching vectorizer\nif \"Char\" in best_name:\n    joblib.dump(tfidf_char, \"tfidf_best.pkl\")\nelse:\n    joblib.dump(tfidf_word, \"tfidf_best.pkl\")\n\njoblib.dump(best_model, \"best_model.pkl\")\n\nprint(\"Saved: tfidf_best.pkl and best_model.pkl in /kaggle/working\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:20:43.220234Z","iopub.execute_input":"2025-12-15T22:20:43.220656Z","iopub.status.idle":"2025-12-15T22:20:43.491728Z","shell.execute_reply.started":"2025-12-15T22:20:43.220627Z","shell.execute_reply":"2025-12-15T22:20:43.490690Z"}},"outputs":[{"name":"stdout","text":"Best model: Random Forest (Word TF-IDF) | F1: 0.8938156359393232 | AUC: 0.8680312499999999\nSaved: tfidf_best.pkl and best_model.pkl in /kaggle/working\n","output_type":"stream"}],"execution_count":8}]}