import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler, LabelEncoder
from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier
from sklearn.ensemble import RandomForestClassifier
from sklearn.svm import SVC
from sklearn.naive_bayes import GaussianNB
from sklearn.neighbors import KNeighborsClassifier
from sklearn.datasets import load_iris
from sklearn.feature_selection import SelectKBest, chi2, RFE
from sklearn.metrics import (
confusion_matrix,
accuracy_score,
precision_score,
recall_score,
f1_score,
classification_report
)
from sklearn.model_selection import GridSearchCV


# Step 1 - laod dataset

df = pd.read_csv('Heart_Disease_Prediction.csv')

# Step 2 - Handle missing values

print(df.isnull().sum())
 
df['Heart Disease'] = df['Heart Disease'].fillna(df['Heart Disease'].mode()[0])

print(df.isnull().sum())

# Step 3 - Apply machine learning models with a grid search optimizer to the cleaned datasets and evaluate the models' performance

X = df.drop('Heart Disease', axis=1).values
y = df['Heart Disease'].values

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

def evaluate_model(model_name, y_true, y_pred):
    accuracy = accuracy_score(y_true, y_pred)
    precision = precision_score(y_true, y_pred, average='weighted')
    cm = confusion_matrix(y_true, y_pred)
    
    # Create a report
    report = classification_report(y_true, y_pred)
    
    # Output results
    metrics = {
        'Model Name': model_name,
        'Accuracy': accuracy,
        'Precision': precision,
        'Classification Report': report
    }
    
    # Plot Confusion Matrix
    plt.figure(figsize=(4, 4))
    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', cbar=False,
                xticklabels=np.unique(y_true), yticklabels=np.unique(y_true))
    plt.title(f'Confusion Matrix for {model_name}')
    plt.xlabel('Predicted Label')
    plt.ylabel('True Label')
    plt.show()
    
    return metrics
    
#Random Forest Classifier Model

rf_classifier = RandomForestClassifier(random_state=42)

# Define the hyperparameters for grid search
param_grid = {
'n_estimators': [100, 200, 300],
'max_depth': [None, 10, 20, 30],
'min_samples_split': [2, 5, 10],
'min_samples_leaf': [1, 2, 4]
}
# Initialize GridSearchCV
grid_search = GridSearchCV(estimator=rf_classifier, param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)

# Train the model
rf_classifier.fit(X_train, y_train)

# Make predictions
y_pred = rf_classifier.predict(X_test)

# Evaluate the model
evaluation_results = evaluate_model('RandomForestClassifier', y_test, y_pred)

# Print the evaluation results
for key, value in evaluation_results.items():
    if key == 'Classification Report':
        print(value)  # Print report separately for better readability
    else:
        # Format the float values to 4 decimal places, or print the value directly if it's not a float
        if isinstance(value, float):
            print(f"{key}: {value:.4f}")
        else:
            print(f"{key}: \n{value}")
            
# LogisticRegression Model
            
LR = LogisticRegression()
# Train the model
LR.fit(X_train, y_train)
# Make predictions
y_pred = LR.predict(X_test)
# Evaluate the model
evaluation_results = evaluate_model('LogisticRegression', y_test, y_pred)
# Print the evaluation results
# Print the evaluation results
for key, value in evaluation_results.items():
    if key == 'Classification Report':
        print(value)  # Print report separately for better readability
    else:
        # Format the float values to 4 decimal places, or print the value directly if it's not a float
        if isinstance(value, float):
            print(f"{key}: {value:.4f}")
        else:
            print(f"{key}: \n{value}")
            
            

# Step 4 - Feature selection

select_feature = SelectKBest(chi2,k=8).fit(X_train, y_train)
data = pd.DataFrame({
    'Feature': df.drop('Heart Disease', axis=1).columns,  # Use the feature names from the original DataFrame
    'Score': select_feature.scores_
})
X_train_selected = select_feature.transform(X_train)
X_test_selected = select_feature.transform(X_test)

print(data)

# Transform the training and testing data using the selected features
X_train_selected = select_feature.transform(X_train)
X_test_selected = select_feature.transform(X_test)

# Step 5 - Applying RF on selected features

def evaluate_model(model_name, y_true, y_pred):
# Calculate metrics
    accuracy = accuracy_score(y_true, y_pred)
    precision = precision_score(y_true, y_pred, average='weighted')
    recall = recall_score(y_true, y_pred, average='weighted')
    f1 = f1_score(y_true, y_pred, average='weighted')
    # Output results
    metrics = {
     'Model Name': model_name,
     'Accuracy': accuracy,
     'Precision': precision,
     'Recall': recall,
     'F1 Score': f1,
    }
    return metrics
    
# Define the Random Forest Classifier

rf_classifier = RandomForestClassifier(random_state=42)

# Define the hyperparameters for grid search
param_grid = {
'n_estimators': [100, 200, 300],
'max_depth': [None, 10, 20, 30],
'min_samples_split': [2, 5, 10],
'min_samples_leaf': [1, 2, 4]
}
# Initialize GridSearchCV
grid_search = GridSearchCV(estimator=rf_classifier,
param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)
# Fit the grid search to the data
grid_search.fit(X_train_selected, y_train)
# Get the best estimator from grid search
best_rf_classifier = grid_search.best_estimator_
# Make predictions using the best model
y_pred = best_rf_classifier.predict(X_test_selected)
# Evaluate the model
evaluation_results = evaluate_model('RandomForestClassifier', y_test,
y_pred)

# Print the evaluation results
for key, value in evaluation_results.items():
    print(f"{key}: {value:.4f}" 
if isinstance(value, float) else
f"{key}: \n{value}")
# Print the best parameters found by grid search
print("\nBest hyperparameters found by GridSearchCV:")
print(grid_search.best_params_)