# Fahad AbuBaker Bahashwan
# Assigment-2

# Import required libraries
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns

# Load the dataset
df = pd.read_csv("Housing.csv")

# Show first 5 rows
print("First 5 rows of the data:")
print(df.head())

# =====================================
# 1. Handling Missing Values
# =====================================

# Check for missing values
print("\nMissing values in each column:")
print(df.isnull().sum())

# Since this dataset has no missing values, we don't need to fill or drop anything

# =====================================
# 2. Handling Outliers
# =====================================

# We will use the IQR method to find and remove outliers
def remove_outliers(data, column):
    Q1 = data[column].quantile(0.25)
    Q3 = data[column].quantile(0.75)
    IQR = Q3 - Q1
    lower_limit = Q1 - 1.5 * IQR
    upper_limit = Q3 + 1.5 * IQR
    # Keep only data within the limit
    return data[(data[column] >= lower_limit) & (data[column] <= upper_limit)]

# Columns we want to clean
columns_to_check = ['price', 'area', 'bedrooms', 'bathrooms', 'stories', 'parking']

# Remove outliers from each column
for col in columns_to_check:
    df = remove_outliers(df, col)

print(f"\nData shape after removing outliers: {df.shape}")

# =====================================
# 3. Data Visualization
# =====================================

# Set the style
sns.set(style="whitegrid")

# Plot histograms for numerical columns
for col in columns_to_check:
    plt.figure(figsize=(6, 4))
    sns.histplot(df[col], kde=True)
    plt.title(f"{col} Distribution")
    plt.xlabel(col)
    plt.ylabel("Frequency")
    plt.show()

# Show correlation heatmap
plt.figure(figsize=(8, 6))
sns.heatmap(df[columns_to_check].corr(), annot=True, cmap='coolwarm')
plt.title("Correlation Heatmap")
plt.show()
