Dogs vs Cats Classification -Using KAGGLE DATASETΒΆ

Downloaded Kaggle Dataset https://www.kaggle.com/datasets/salader/dogs-vs-catsΒΆ

InΒ [1]:
from tensorflow.python.client import device_lib
print(device_lib.list_local_devices())
      
[name: "/device:CPU:0"
device_type: "CPU"
memory_limit: 268435456
locality {
}
incarnation: 16825190368384564054
xla_global_id: -1
]
InΒ [11]:
# ============================================================================
# STEP 1: IMPORT LIBRARIES AND SETUP
# ============================================================================

import os
import shutil
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from pathlib import Path
import random
from PIL import Image

# Deep Learning Libraries
import tensorflow as tf
from tensorflow import keras
from tensorflow.keras import layers
from tensorflow.keras.preprocessing.image import ImageDataGenerator
from tensorflow.keras.applications import VGG16
from tensorflow.keras.optimizers import Adam
from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint

# Machine Learning Utilities
from sklearn.metrics import classification_report, confusion_matrix, accuracy_score
from sklearn.metrics import precision_score, recall_score, f1_score, roc_curve, auc

print("βœ… All libraries imported successfully!")
print(f"πŸ”₯ TensorFlow version: {tf.__version__}")
print(f"🐍 Using GPU: {len(tf.config.list_physical_devices('GPU')) > 0}")

# Set random seeds for reproducibility
random.seed(42)
np.random.seed(42)
tf.random.set_seed(42)
βœ… All libraries imported successfully!
πŸ”₯ TensorFlow version: 2.19.0
🐍 Using GPU: False
InΒ [12]:
from pathlib import Path

# ============================================================================
# STEP 2: DETECT AND ORGANIZE YOUR REAL DATASET
# ============================================================================

def detect_dataset_structure():
    """
    Automatically detect what structure the dataset in './dataset' has.
    """
    print("πŸ” Looking for Kaggle Dogs vs Cats dataset...")

    # Fixed dataset location
    path = Path('./dataset')
    dataset_path = None
    dataset_type = None

    if path.exists():
        print(f"πŸ“ Found folder: {path}")
        
        # Check for original Kaggle structure (train folder with mixed images)
        train_folder = path / 'train'
        if train_folder.exists():
            image_files = list(train_folder.glob('*.jpg'))
            if len(image_files) > 100:  # Should have thousands of images
                dataset_path = path
                dataset_type = 'kaggle_original'
                print(f"βœ… Found Kaggle original structure with {len(image_files)} images!")
        
        # Check for organized structure (separate cat/dog folders)
        if dataset_path is None:
            cats_folder = path / 'cats'
            dogs_folder = path / 'dogs'
            if cats_folder.exists() and dogs_folder.exists():
                cat_images = list(cats_folder.glob('*.jpg'))
                dog_images = list(dogs_folder.glob('*.jpg'))
                if len(cat_images) > 50 and len(dog_images) > 50:
                    dataset_path = path
                    dataset_type = 'organized'
                    print(f"βœ… Found organized structure: {len(cat_images)} cats, {len(dog_images)} dogs!")

        # Check if this folder contains image files directly
        if dataset_path is None:
            image_files = list(path.glob('*.jpg'))
            if len(image_files) > 100:
                dataset_path = path
                dataset_type = 'flat'
                print(f"βœ… Found flat structure with {len(image_files)} images!")

    if dataset_path is None:
        print("❌ Could not find a valid dataset structure in './dataset'!")
        print("\nπŸ“‹ Please make sure your dataset folder contains:")
        print("   - 'train' folder with images (Kaggle original), OR")
        print("   - 'cats/' and 'dogs/' folders (organized), OR")
        print("   - Many JPG images directly (flat).")
        return None, None
    
    return dataset_path, dataset_type

# Detect your dataset
DATASET_PATH, DATASET_TYPE = detect_dataset_structure()

if DATASET_PATH:
    print(f"\nπŸŽ‰ Dataset found at: {DATASET_PATH.absolute()}")
    print(f"πŸ“Š Dataset type: {DATASET_TYPE}")
else:
    print("\n⚠️  Please place your Kaggle dataset in the 'dataset' folder and run this cell again.")
πŸ” Looking for Kaggle Dogs vs Cats dataset...
πŸ“ Found folder: dataset
βœ… Found organized structure: 10000 cats, 10000 dogs!

πŸŽ‰ Dataset found at: D:\MidOcean\Practical Vision\CV_Assignment1\dataset
πŸ“Š Dataset type: organized
InΒ [10]:
# ============================================================================
# STEP 3: ORGANIZE DATASET
# ============================================================================

def organize_kaggle_dataset(dataset_path, dataset_type):
    """
    Organize your Kaggle dataset into proper train/validation/test structure.
    """
    if dataset_path is None:
        print("❌ No dataset found to organize!")
        return None
    
    print(f"πŸ”§ Organizing your {dataset_type} dataset...")
    
    # Create organized dataset folder
    organized_path = Path('./organized_dataset')
    organized_path.mkdir(exist_ok=True)
    
    # Create train/validation/test folders
    for split in ['train', 'validation', 'test']:
        for class_name in ['cats', 'dogs']:
            (organized_path / split / class_name).mkdir(parents=True, exist_ok=True)
    
    # βœ… Move split_images OUTSIDE the if-statement
    def split_images(images, class_name):
        random.shuffle(images)  # Shuffle for random split

        n_total = len(images)
        n_train = int(0.7 * n_total)
        n_val = int(0.2 * n_total)

        train_imgs = images[:n_train]
        val_imgs = images[n_train:n_train + n_val]
        test_imgs = images[n_train + n_val:]

        splits = {'train': train_imgs, 'validation': val_imgs, 'test': test_imgs}

        for split_name, img_list in splits.items():
            dest_folder = organized_path / split_name / class_name
            for i, img_path in enumerate(img_list):
                dest_path = dest_folder / f"{class_name[:-1]}_{i:04d}.jpg"
                if not dest_path.exists():
                    shutil.copy2(img_path, dest_path)
            print(f"   πŸ“ {split_name}/{class_name}: {len(img_list)} images")
    
    # ==============================
    # Process based on dataset type
    # ==============================
    if dataset_type == 'kaggle_original':
        train_folder = dataset_path / 'train'
        cat_images = sorted(list(train_folder.glob('cat.*.jpg')))
        dog_images = sorted(list(train_folder.glob('dog.*.jpg')))

        print(f"πŸ“Š Found {len(cat_images)} cat images and {len(dog_images)} dog images")

        print("\n🐱 Organizing cat images...")
        split_images(cat_images, 'cats')

        print("\n🐢 Organizing dog images...")
        split_images(dog_images, 'dogs')

    elif dataset_type == 'organized':
        print("βœ… Dataset is already organized! Creating train/val/test splits...")

        for class_name in ['cats', 'dogs']:
            class_folder = dataset_path / class_name
            images = list(class_folder.glob('*.jpg'))

            print(f"\nπŸ“Š Processing {len(images)} {class_name} images...")
            split_images(images, class_name)

    print(f"\nβœ… Dataset organized successfully!")
    print(f"πŸ“ Organized dataset location: {organized_path.absolute()}")

    return organized_path


# Organize your dataset
if DATASET_PATH:
    ORGANIZED_PATH = organize_kaggle_dataset(DATASET_PATH, DATASET_TYPE)

    if ORGANIZED_PATH:
        print("\n🎯 Dataset is ready for training!")

        # Show final structure
        print("\nπŸ“Š Final Dataset Structure:")
        for split in ['train', 'validation', 'test']:
            for class_name in ['cats', 'dogs']:
                folder = ORGANIZED_PATH / split / class_name
                count = len(list(folder.glob('*.jpg')))
                print(f"   πŸ“ {split}/{class_name}: {count:,} images")
else:
    print("⚠️  Cannot organize dataset - please check dataset location first.")
    ORGANIZED_PATH = None
πŸ”§ Organizing your organized dataset...
βœ… Dataset is already organized! Creating train/val/test splits...

πŸ“Š Processing 10000 cats images...
   πŸ“ train/cats: 7000 images
   πŸ“ validation/cats: 2000 images
   πŸ“ test/cats: 1000 images

πŸ“Š Processing 10000 dogs images...
   πŸ“ train/dogs: 7000 images
   πŸ“ validation/dogs: 2000 images
   πŸ“ test/dogs: 1000 images

βœ… Dataset organized successfully!
πŸ“ Organized dataset location: D:\MidOcean\Practical Vision\CV_Assignment1\organized_dataset

🎯 Dataset is ready for training!

πŸ“Š Final Dataset Structure:
   πŸ“ train/cats: 7,000 images
   πŸ“ train/dogs: 7,000 images
   πŸ“ validation/cats: 2,000 images
   πŸ“ validation/dogs: 2,000 images
   πŸ“ test/cats: 1,000 images
   πŸ“ test/dogs: 1,000 images
InΒ [5]:
# ============================================================================
# STEP 4: VISUALIZE DATASET
# ============================================================================

def show_real_dataset_samples(organized_path):
    """
    Show sample images from your real Kaggle dataset.
    """
    if organized_path is None:
        print("❌ No organized dataset to display!")
        return
    
    print("πŸ–ΌοΈ  Showing samples from your REAL Kaggle dataset...")
    
    fig, axes = plt.subplots(4, 4, figsize=(12, 12))
    fig.suptitle('Kaggle Dogs vs Cats Dataset', fontsize=16, fontweight='bold')
    
    # Show 8 cats and 8 dogs from training set
    cats_folder = organized_path / 'train' / 'cats'
    dogs_folder = organized_path / 'train' / 'dogs'
    
    cat_images = list(cats_folder.glob('*.jpg'))[:8]
    dog_images = list(dogs_folder.glob('*.jpg'))[:8]
    
    # Display cats in first two rows
    for i, img_path in enumerate(cat_images):
        row = i // 4
        col = i % 4
        
        try:
            img = Image.open(img_path)
            axes[row, col].imshow(img)
            axes[row, col].set_title(f'🐱 Cat {i+1}', fontweight='bold')
            axes[row, col].axis('off')
        except Exception as e:
            axes[row, col].text(0.5, 0.5, f'Error loading\n{img_path.name}', 
                               ha='center', va='center', transform=axes[row, col].transAxes)
            axes[row, col].axis('off')
    
    # Display dogs in last two rows
    for i, img_path in enumerate(dog_images):
        row = (i // 4) + 2
        col = i % 4
        
        try:
            img = Image.open(img_path)
            axes[row, col].imshow(img)
            axes[row, col].set_title(f'🐢 Dog {i+1}', fontweight='bold')
            axes[row, col].axis('off')
        except Exception as e:
            axes[row, col].text(0.5, 0.5, f'Error loading\n{img_path.name}', 
                               ha='center', va='center', transform=axes[row, col].transAxes)
            axes[row, col].axis('off')
    
    plt.tight_layout()
    plt.show()
    
InΒ [7]:
# ============================================================================
# STEP 5: CREATE DATA GENERATORS FOR REAL DATASET
# ============================================================================

def create_real_data_generators(organized_path, img_size=150, batch_size=32):
    """
    Create data generators for your real Kaggle dataset with proper augmentation.
    """
    if organized_path is None:
        print("❌ No organized dataset available!")
        return None, None, None
    
    print(f"πŸ”§ Creating data generators for real dataset...")
    print(f"   πŸ“ Image size: {img_size}x{img_size}")
    print(f"   πŸ“¦ Batch size: {batch_size}")
    
    # Training data generator with augmentation
    train_datagen = ImageDataGenerator(
        rescale=1./255,              # Normalize pixel values to 0-1
        rotation_range=20,           # Rotate images up to 20 degrees
        width_shift_range=0.2,       # Shift images horizontally
        height_shift_range=0.2,      # Shift images vertically
        shear_range=0.2,             # Shear transformation
        zoom_range=0.2,              # Zoom in/out
        horizontal_flip=True,        # Flip images horizontally
        fill_mode='nearest'          # Fill missing pixels
    )
    
    # Validation and test data generators (no augmentation, only rescaling)
    val_test_datagen = ImageDataGenerator(rescale=1./255)
    
    # Create generators
    train_generator = train_datagen.flow_from_directory(
        organized_path / 'train',
        target_size=(img_size, img_size),
        batch_size=batch_size,
        class_mode='binary',
        shuffle=True,
        seed=42
    )
    
    validation_generator = val_test_datagen.flow_from_directory(
        organized_path / 'validation',
        target_size=(img_size, img_size),
        batch_size=batch_size,
        class_mode='binary',
        shuffle=False,
        seed=42
    )
    
    test_generator = val_test_datagen.flow_from_directory(
        organized_path / 'test',
        target_size=(img_size, img_size),
        batch_size=batch_size,
        class_mode='binary',
        shuffle=False,
        seed=42
    )
    
    print(f"\nβœ… Data generators created successfully!")
    print(f"   πŸ‹οΈ  Training images: {train_generator.samples:,}")
    print(f"   πŸ§ͺ Validation images: {validation_generator.samples:,}")
    print(f"   🎯 Test images: {test_generator.samples:,}")
    print(f"   πŸ“Š Class indices: {train_generator.class_indices}")
    
    return train_generator, validation_generator, test_generator

# Create data generators for your real dataset
if ORGANIZED_PATH:
    train_gen, val_gen, test_gen = create_real_data_generators(ORGANIZED_PATH)
    
    if train_gen:
        print("\nπŸŽ‰ Ready to train on Kaggle dataset!")
      
    else:
        print("❌ Failed to create data generators")
else:
    print("⚠️  Cannot create data generators - dataset not organized.")
    train_gen = val_gen = test_gen = None
πŸ”§ Creating data generators for real dataset...
   πŸ“ Image size: 150x150
   πŸ“¦ Batch size: 32
Found 14000 images belonging to 2 classes.
Found 4000 images belonging to 2 classes.
Found 2000 images belonging to 2 classes.

βœ… Data generators created successfully!
   πŸ‹οΈ  Training images: 14,000
   πŸ§ͺ Validation images: 4,000
   🎯 Test images: 2,000
   πŸ“Š Class indices: {'cats': 0, 'dogs': 1}

πŸŽ‰ Ready to train on Kaggle dataset!
InΒ [8]:
# ============================================================================
# STEP 6: BUILD CUSTOM CNN MODEL
# ============================================================================

def build_custom_cnn(img_size=150):
    """
    Build a custom CNN model for real cat and dog classification.
    """
    print("Building Custom CNN for REAL dataset...")
    
    model = keras.Sequential([
        # Input layer
        layers.Input(shape=(img_size, img_size, 3)),
        
        # First convolutional block
        layers.Conv2D(32, (3, 3), activation='relu'),
        layers.MaxPooling2D(2, 2),
        
        # Second convolutional block
        layers.Conv2D(64, (3, 3), activation='relu'),
        layers.MaxPooling2D(2, 2),
        
        # Third convolutional block
        layers.Conv2D(128, (3, 3), activation='relu'),
        layers.MaxPooling2D(2, 2),
        
        # Fourth convolutional block
        layers.Conv2D(128, (3, 3), activation='relu'),
        layers.MaxPooling2D(2, 2),
        
        # Flatten and dense layers
        layers.Flatten(),
        layers.Dropout(0.5),  # Prevent overfitting
        layers.Dense(512, activation='relu'),
        layers.Dropout(0.5),
        layers.Dense(1, activation='sigmoid')  # Binary classification
    ])
    
    # Compile the model
    model.compile(
        optimizer=Adam(learning_rate=0.001),
        loss='binary_crossentropy',
        metrics=['accuracy']
    )
    
    print(" Custom CNN built successfully!")
    print(" This model will learn cat and dog features from scratch!")
    
    return model

# Build custom CNN
if train_gen:
    custom_model = build_custom_cnn()
    custom_model.summary()
else:
    print("⚠️  Cannot build model - data generators not available.")
    custom_model = None
Building Custom CNN for REAL dataset...
 Custom CNN built successfully!
 This model will learn cat and dog features from scratch!
Model: "sequential"
┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓
┃ Layer (type)                         ┃ Output Shape                ┃         Param # ┃
┑━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩
β”‚ conv2d (Conv2D)                      β”‚ (None, 148, 148, 32)        β”‚             896 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ max_pooling2d (MaxPooling2D)         β”‚ (None, 74, 74, 32)          β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ conv2d_1 (Conv2D)                    β”‚ (None, 72, 72, 64)          β”‚          18,496 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ max_pooling2d_1 (MaxPooling2D)       β”‚ (None, 36, 36, 64)          β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ conv2d_2 (Conv2D)                    β”‚ (None, 34, 34, 128)         β”‚          73,856 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ max_pooling2d_2 (MaxPooling2D)       β”‚ (None, 17, 17, 128)         β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ conv2d_3 (Conv2D)                    β”‚ (None, 15, 15, 128)         β”‚         147,584 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ max_pooling2d_3 (MaxPooling2D)       β”‚ (None, 7, 7, 128)           β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ flatten (Flatten)                    β”‚ (None, 6272)                β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ dropout (Dropout)                    β”‚ (None, 6272)                β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ dense (Dense)                        β”‚ (None, 512)                 β”‚       3,211,776 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ dropout_1 (Dropout)                  β”‚ (None, 512)                 β”‚               0 β”‚
β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
β”‚ dense_1 (Dense)                      β”‚ (None, 1)                   β”‚             513 β”‚
β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
 Total params: 3,453,121 (13.17 MB)
 Trainable params: 3,453,121 (13.17 MB)
 Non-trainable params: 0 (0.00 B)
InΒ [9]:
# ============================================================================
# STEP 7: TRAIN CUSTOM CNN ON REAL DATA
# ============================================================================

def train_on_real_data(model, model_name, train_gen, val_gen, epochs=10):
    """
    Train model on Kaggle dataset.
    """
    print(f"\nπŸ‹οΈ  Training {model_name} on REAL Kaggle data...")
    print("=" * 60)
    print("cat and dog photos!")
    print(f"πŸ“Š Training on {train_gen.samples:,} real images")
    print(f"πŸ§ͺ Validating on {val_gen.samples:,} real images")
    
    # Create callbacks
    callbacks = [
        EarlyStopping(
            monitor='val_loss',
            patience=5,
            restore_best_weights=True,
            verbose=1
        ),
        ReduceLROnPlateau(
            monitor='val_loss',
            factor=0.2,
            patience=3,
            min_lr=1e-7,
            verbose=1
        ),
        ModelCheckpoint(
            filepath=f'best_{model_name.lower().replace(" ", "_")}_real.h5',
            monitor='val_accuracy',
            save_best_only=True,
            verbose=1
        )
    ]
    
    print(f"\nπŸš€ Starting training (this may take a while with real data)...")
    
    # Train the model
    history = model.fit(
        train_gen,
        epochs=epochs,
        validation_data=val_gen,
        callbacks=callbacks,
        verbose=1
    )
    
    print(f"\nβœ… {model_name} training completed on REAL data!")
    
    # Show final results
    final_train_acc = history.history['accuracy'][-1]
    final_val_acc = history.history['val_accuracy'][-1]
    
    print(f"\n Final Results on REAL Data:")
    print(f"   Training accuracy: {final_train_acc:.1%}")
    print(f"   Validation accuracy: {final_val_acc:.1%}")
    
    if final_val_acc > 0.85:
        print(f"    Excellent! AI learned real cat/dog features very well!")
    elif final_val_acc > 0.75:
        print(f"    Good! AI learned to distinguish real cats from dogs!")
    else:
        print(f"    AI is learning, but real data is challenging!")
    
    return history

# Train custom CNN on real data
if custom_model and train_gen and val_gen:
    print("🎯 Training Custom CNN on Kaggle dataset...")
    custom_history = train_on_real_data(custom_model, "Custom CNN", train_gen, val_gen)
else:
    print("⚠️  Cannot train - model or data not available.")
    custom_history = None
🎯 Training Custom CNN on Kaggle dataset...

πŸ‹οΈ  Training Custom CNN on REAL Kaggle data...
============================================================
cat and dog photos!
πŸ“Š Training on 14,000 real images
πŸ§ͺ Validating on 4,000 real images

πŸš€ Starting training (this may take a while with real data)...
C:\Programs\python\lib\site-packages\keras\src\trainers\data_adapters\py_dataset_adapter.py:121: UserWarning: Your `PyDataset` class should call `super().__init__(**kwargs)` in its constructor. `**kwargs` can include `workers`, `use_multiprocessing`, `max_queue_size`. Do not pass these arguments to `fit()`, as they will be ignored.
  self._warn_if_super_not_called()
Epoch 1/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.5359 - loss: 0.6868
Epoch 1: val_accuracy improved from -inf to 0.62350, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 734s 2s/step - accuracy: 0.5360 - loss: 0.6868 - val_accuracy: 0.6235 - val_loss: 0.6469 - learning_rate: 0.0010
Epoch 2/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 2s/step - accuracy: 0.6333 - loss: 0.6423
Epoch 2: val_accuracy improved from 0.62350 to 0.69900, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 1151s 3s/step - accuracy: 0.6334 - loss: 0.6423 - val_accuracy: 0.6990 - val_loss: 0.5838 - learning_rate: 0.0010
Epoch 3/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.6781 - loss: 0.5969
Epoch 3: val_accuracy improved from 0.69900 to 0.74450, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 722s 2s/step - accuracy: 0.6782 - loss: 0.5969 - val_accuracy: 0.7445 - val_loss: 0.5144 - learning_rate: 0.0010
Epoch 4/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 4s/step - accuracy: 0.7058 - loss: 0.5643
Epoch 4: val_accuracy did not improve from 0.74450
438/438 ━━━━━━━━━━━━━━━━━━━━ 1700s 4s/step - accuracy: 0.7058 - loss: 0.5643 - val_accuracy: 0.7072 - val_loss: 0.5428 - learning_rate: 0.0010
Epoch 5/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.7374 - loss: 0.5284
Epoch 5: val_accuracy did not improve from 0.74450
438/438 ━━━━━━━━━━━━━━━━━━━━ 736s 2s/step - accuracy: 0.7374 - loss: 0.5284 - val_accuracy: 0.7295 - val_loss: 0.5182 - learning_rate: 0.0010
Epoch 6/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.7512 - loss: 0.5096
Epoch 6: val_accuracy improved from 0.74450 to 0.78925, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 767s 2s/step - accuracy: 0.7512 - loss: 0.5096 - val_accuracy: 0.7893 - val_loss: 0.4594 - learning_rate: 0.0010
Epoch 7/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.7608 - loss: 0.4999
Epoch 7: val_accuracy improved from 0.78925 to 0.83275, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 761s 2s/step - accuracy: 0.7608 - loss: 0.4999 - val_accuracy: 0.8328 - val_loss: 0.3879 - learning_rate: 0.0010
Epoch 8/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.7801 - loss: 0.4641
Epoch 8: val_accuracy improved from 0.83275 to 0.83450, saving model to best_custom_cnn_real.h5
WARNING:absl:You are saving your model as an HDF5 file via `model.save()` or `keras.saving.save_model(model)`. This file format is considered legacy. We recommend using instead the native Keras format, e.g. `model.save('my_model.keras')` or `keras.saving.save_model(model, 'my_model.keras')`. 
438/438 ━━━━━━━━━━━━━━━━━━━━ 697s 2s/step - accuracy: 0.7801 - loss: 0.4641 - val_accuracy: 0.8345 - val_loss: 0.3811 - learning_rate: 0.0010
Epoch 9/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 2s/step - accuracy: 0.7961 - loss: 0.4558
Epoch 9: val_accuracy did not improve from 0.83450
438/438 ━━━━━━━━━━━━━━━━━━━━ 1227s 3s/step - accuracy: 0.7961 - loss: 0.4557 - val_accuracy: 0.8295 - val_loss: 0.3851 - learning_rate: 0.0010
Epoch 10/10
438/438 ━━━━━━━━━━━━━━━━━━━━ 0s 1s/step - accuracy: 0.8002 - loss: 0.4361
Epoch 10: val_accuracy did not improve from 0.83450
438/438 ━━━━━━━━━━━━━━━━━━━━ 715s 2s/step - accuracy: 0.8002 - loss: 0.4361 - val_accuracy: 0.7790 - val_loss: 0.4413 - learning_rate: 0.0010
Restoring model weights from the end of the best epoch: 8.

βœ… Custom CNN training completed on REAL data!

 Final Results on REAL Data:
   Training accuracy: 80.6%
   Validation accuracy: 77.9%
    Good! AI learned to distinguish real cats from dogs!