import pandas as pd
from sklearn.feature_selection import SelectKBest, chi2, RFE
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split, GridSearchCV
from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
)

# Load the dataset
data = pd.read_csv("heart_disease.csv")

# Split the data into features and target
X = data.drop("target", axis=1)
y = data["target"]

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

# Feature selection using SelectKBest
select_feature = SelectKBest(chi2, k=8).fit(X_train, y_train)
X_train_selected = select_feature.transform(X_train)
X_test_selected = select_feature.transform(X_test)

# Function to evaluate the model
def evaluate_model(model_name, y_true, y_pred):
    accuracy = accuracy_score(y_true, y_pred)
    precision = precision_score(y_true, y_pred, average="weighted")
    recall = recall_score(y_true, y_pred, average="weighted")
    f1 = f1_score(y_true, y_pred, average="weighted")
    metrics = {
        "Model Name": model_name,
        "Accuracy": accuracy,
        "Precision": precision,
        "Recall": recall,
        "F1 Score": f1,
    }
    return metrics

if __name__ == "__main__":
    print("Select KBest:")
    rf_classifier = RandomForestClassifier(random_state=42)
    param_grid = {
        "n_estimators": [100, 200, 300],
        "max_depth": [None, 10, 20, 30],
        "min_samples_split": [2, 5, 10],
        "min_samples_leaf": [1, 2, 4],
    }
    grid_search = GridSearchCV(
        estimator=rf_classifier,
        param_grid=param_grid,
        cv=5,
        scoring="accuracy",
        n_jobs=-1,
    )
    grid_search.fit(X_train_selected, y_train)
    best_rf_classifier = grid_search.best_estimator_
    y_pred = best_rf_classifier.predict(X_test_selected)
    evaluation_results = evaluate_model("RandomForestClassifier", y_test, y_pred)
    for key, value in evaluation_results.items():
        print(
            f"{key}: {value:.4f}" if isinstance(value, float) else f"{key}: \n{value}"
        )
    print("\nBest hyperparameters found by GridSearchCV:")
    print(grid_search.best_params_)
    
    print("\RFE (Wrapper Method):")
    clf_rf_2 = RandomForestClassifier(random_state=43)
    rfe_selector = RFE(estimator=clf_rf_2, n_features_to_select=8, step=1)
    X_train_selected_rfe = rfe_selector.fit_transform(X_train, y_train)
    X_test_selected_rfe = rfe_selector.transform(X_test)
    selected_features = rfe_selector.get_support(indices=True)
    print("Selected Features (RFE):", selected_features)
    rf_classifier = RandomForestClassifier(random_state=42)
    param_grid = {
        "n_estimators": [100, 200, 300],
        "max_depth": [None, 10, 20, 30],
        "min_samples_split": [2, 5, 10],
        "min_samples_leaf": [1, 2, 4],
    }
    grid_search = GridSearchCV(
        estimator=rf_classifier,
        param_grid=param_grid,
        cv=5,
        scoring="accuracy",
        n_jobs=-1,
    )
    grid_search.fit(X_train_selected_rfe, y_train)
    best_rf_classifier = grid_search.best_estimator_
    y_pred = best_rf_classifier.predict(X_test_selected_rfe)
    evaluation_results = evaluate_model("RandomForestClassifier", y_test, y_pred)
    for key, value in evaluation_results.items():
        print(
            f"{key}: {value:.4f}" if isinstance(value, float) else f"{key}: \n{value}"
        )
    print("\nBest hyperparameters found by GridSearchCV:")
    print(grid_search.best_params_)