import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.svm import SVC
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import GridSearchCV
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LogisticRegression
from sklearn.feature_selection import SelectKBest, chi2
import matplotlib.pyplot as plt

# Load the dataset
file_path = r"C:\Users\faljaroudi\Desktop\Master\ML Project\diabetes_012_health_indicators_BRFSS2015_cleaned.csv"
diabetes_data = pd.read_csv(file_path)
print(f"Dataset loaded with {diabetes_data.shape[0]} rows and {diabetes_data.shape[1]} columns.")
print(diabetes_data.head())

# Split into features and target
X = diabetes_data.drop('Diabetes_012', axis=1)
y = diabetes_data['Diabetes_012']
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)

# Feature Selection with Chi-Square
X_train_chi2 = X_train.apply(pd.to_numeric, errors='coerce').fillna(0)
chi2_selector = SelectKBest(chi2, k=5)
X_train_chi2_selected = chi2_selector.fit_transform(X_train_chi2, y_train)
chi2_scores = chi2_selector.scores_
chi2_feature_scores = pd.DataFrame({'Feature': X_train.columns, 'Chi2 Score': chi2_scores})
chi2_feature_scores = chi2_feature_scores.sort_values(by='Chi2 Score', ascending=False)
print("Chi-Square Feature Scores:")
print(chi2_feature_scores)

plt.figure(figsize=(10, 6))
plt.barh(chi2_feature_scores['Feature'], chi2_feature_scores['Chi2 Score'])
plt.title('Feature Importance from Chi-Square Test')
plt.xlabel('Chi-Square Score')
plt.ylabel('Feature')
plt.show()

# Feature Selection with Random Forest
rf_model = RandomForestClassifier(random_state=42)
rf_model.fit(X_train, y_train)
rf_feature_importances = rf_model.feature_importances_
rf_feature_scores = pd.DataFrame({'Feature': X_train.columns, 'Importance': rf_feature_importances})
rf_feature_scores = rf_feature_scores.sort_values(by='Importance', ascending=False)
print("\nRandom Forest Feature Importance Scores:")
print(rf_feature_scores)

plt.figure(figsize=(10, 6))
plt.barh(rf_feature_scores['Feature'], rf_feature_scores['Importance'])
plt.title('Feature Importance from Random Forest')
plt.xlabel('Importance')
plt.ylabel('Feature')
plt.show()

# Train Logistic Regression
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train_chi2_selected)
X_test_scaled = scaler.transform(X_test_chi2_selected)
log_reg = LogisticRegression(max_iter=200, random_state=42)
log_reg.fit(X_train_scaled, y_train)
log_reg_pred = log_reg.predict(X_test_scaled)
from sklearn.metrics import classification_report
print("Logistic Regression Classification Report:")
print(classification_report(y_test, log_reg_pred))

# Hyperparameter Tuning for SVM
svm_params = {'C': [1], 'kernel': ['linear'], 'gamma': ['scale']}
svm_model = SVC(random_state=42)
grid_search_svm = GridSearchCV(svm_model, svm_params, scoring='accuracy', cv=5, n_jobs=-1, verbose=3)
print("Starting Grid Search for SVM...")
grid_search_svm.fit(X_train, y_train)
print("Grid Search Completed.")
print(f"Best parameters for SVM: {grid_search_svm.best_params_}")
