{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"cells":[{"cell_type":"markdown","source":"# Competition: cassava-leaf-disease-classification\n\n**Generated by Alexandria Research Assistant**\n\n**Dataset:** cassava-leaf-disease-classification\n\n**Task:** competition - kaggle-competition\n\n---\n\n⚠️ **Note:** This notebook contains Alexandria markers (lines starting with `# ⚠️ ALEXANDRIA MARKER`) at the top of each code cell. These markers enable the 'Sync from Kaggle' feature to track cell outputs. Please do not delete them.","metadata":{}},{"cell_type":"markdown","source":"## Setup & Imports\n\nInstall and import all necessary libraries for image classification, data handling, and visualization.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_1_START===\")\n\n# Setup & Imports\n# Install required libraries\nprint(\"Installing albumentations...\")\n!pip install -U albumentations\n\n# Import necessary libraries\nimport torch\nimport torchvision\nimport albumentations as A\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport random\nfrom PIL import Image\n\n# Set random seeds for reproducibility\nseed = 42\nrandom.seed(seed)\nnp.random.seed(seed)\ntorch.manual_seed(seed)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\ntorch.backends.cudnn.deterministic = True\ntorch.backends.cudnn.benchmark = False\n\n# Define data path\nDATA_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\nprint(f\"Data path set to: {DATA_PATH}\")\n\n# Verify dataset directory exists\nif os.path.exists(DATA_PATH):\n    print(\"Dataset directory found.\")\n    print(\"Contents:\", os.listdir(DATA_PATH))\nelse:\n    print(\"Dataset directory not found. Please check the path.\")\n\n# Print library versions for reproducibility\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"Torchvision version: {torchvision.__version__}\")\nprint(f\"Albumentations version: {A.__version__}\")\nprint(f\"Pandas version: {pd.__version__}\")\nprint(f\"NumPy version: {np.__version__}\")\nprint(f\"Matplotlib version: {plt.matplotlib.__version__}\")\nprint(f\"Seaborn version: {sns.__version__}\")\n\nprint(\"Setup complete.\")","outputs":[],"cell_number":1,"version":1,"status":"generated","created_at":"2025-11-16T12:33:17.779745+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Data Loading\n\nLoad the cassava leaf disease dataset, including image paths and labels, into memory.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_2_START===\")\n\n# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_2_START===\")\n\n# Data Loading Cell\n# Load the cassava leaf disease dataset, including image paths and labels\n\n# Define data path\nDATA_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\n\n# Read train.csv to get image IDs and labels\ntrain_csv_path = os.path.join(DATA_PATH, \"train.csv\")\ndf_train = pd.read_csv(train_csv_path)\n\nprint(f\"Train dataset shape: {df_train.shape}\")\nprint(f\"\\nFirst few rows of train.csv:\")\nprint(df_train.head(10))\n\n# Check for missing values\nprint(f\"\\nMissing values in train dataset:\")\nprint(df_train.isnull().sum())\n\n# Display data types\nprint(f\"\\nData types:\")\nprint(df_train.dtypes)\n\n# Get unique labels and their counts\nprint(f\"\\nLabel distribution:\")\nprint(df_train['label'].value_counts().sort_index())\n\n# Construct full image paths\ndf_train['image_path'] = df_train['image_id'].apply(\n    lambda x: os.path.join(DATA_PATH, 'train_images', f'{x}.jpg')\n)\n\nprint(f\"\\nSample image paths:\")\nprint(df_train['image_path'].head())\n\n# Verify image files exist\nexisting_images = df_train['image_path'].apply(os.path.exists).sum()\ntotal_images = len(df_train)\nprint(f\"\\nImage files found: {existing_images}/{total_images}\")\n\n# Load label mapping if available\nlabel_mapping_path = os.path.join(DATA_PATH, \"label_num_to_disease_map.json\")\nif os.path.exists(label_mapping_path):\n    import json\n    with open(label_mapping_path, 'r') as f:\n        label_mapping = json.load(f)\n    print(f\"\\nLabel mapping loaded:\")\n    print(label_mapping)\n    \n    # Add disease name column to dataframe\n    df_train['disease_name'] = df_train['label'].map(\n        {int(k): v for k, v in label_mapping.items()}\n    )\n    print(f\"\\nDataframe with disease names:\")\n    print(df_train.head())\nelse:\n    print(f\"\\nLabel mapping file not found at {label_mapping_path}\")\n    # Create default label mapping based on observed labels\n    unique_labels = sorted(df_train['label'].unique())\n    label_mapping = {str(i): f\"Class_{i}\" for i in unique_labels}\n    df_train['disease_name'] = df_train['label'].map(\n        {int(k): v for k, v in label_mapping.items()}\n    )\n    print(f\"Using default label mapping: {label_mapping}\")\n\n# Display final dataframe structure\nprint(f\"\\nFinal dataframe structure:\")\nprint(df_train.info())\n\nprint(f\"\\nDataframe preview:\")\nprint(df_train.head(15))\n\n# Summary statistics\nprint(f\"\\nDataset Summary:\")\nprint(f\"Total samples: {len(df_train)}\")\nprint(f\"Number of classes: {df_train['label'].nunique()}\")\nprint(f\"Image directory: {os.path.join(DATA_PATH, 'train_images')}\")\nprint(f\"Images directory exists: {os.path.exists(os.path.join(DATA_PATH, 'train_images'))}\")\n\nprint(\"\\n===ALEXANDRIA_CELL_2_END===\")","outputs":[],"cell_number":2,"version":1,"status":"generated","created_at":"2025-11-16T12:33:25.718477+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Exploratory Data Analysis (EDA)\n\nExplore the dataset to understand class distribution, image samples, and potential data issues.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_3_START===\")\n\n# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_3_START===\")\n\n# Exploratory Data Analysis (EDA)\nprint(\"Starting Exploratory Data Analysis...\")\n\n# Plot class distribution to visualize imbalance\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df_train, x='label', order=sorted(df_train['label'].unique()))\nplt.title('Class Distribution in Cassava Leaf Disease Dataset')\nplt.xlabel('Class Label')\nplt.ylabel('Number of Images')\nplt.xticks(rotation=45)\nplt.show()\n\n# Display random image samples from each class\nnum_samples_per_class = 3\nfig, axes = plt.subplots(len(df_train['label'].unique()), num_samples_per_class, figsize=(12, 10))\naxes = axes if len(df_train['label'].unique()) > 1 else axes.reshape(1, -1)\n\nfor i, label in enumerate(sorted(df_train['label'].unique())):\n    class_df = df_train[df_train['label'] == label]\n    sample_images = class_df.sample(n=min(num_samples_per_class, len(class_df)), random_state=seed)\n    \n    for j, (_, row) in enumerate(sample_images.iterrows()):\n        img_path = row['image_path']\n        if os.path.exists(img_path):\n            img = Image.open(img_path)\n            axes[i, j].imshow(img)\n            axes[i, j].set_title(f\"Class {label}\")\n            axes[i, j].axis('off')\n        else:\n            axes[i, j].text(0.5, 0.5, 'Image not found', transform=axes[i, j].transAxes, ha='center')\n            axes[i, j].axis('off')\n\nplt.tight_layout()\nplt.show()\n\n# Analyze image dimensions and aspect ratios\nprint(\"Analyzing image dimensions and aspect ratios...\")\n\nimage_dims = []\naspect_ratios = []\n\nfor _, row in df_train.head(100).iterrows():  # Limit to first 100 images for speed\n    img_path = row['image_path']\n    if os.path.exists(img_path):\n        with Image.open(img_path) as img:\n            width, height = img.size\n            image_dims.append((width, height))\n            aspect_ratios.append(width / height)\n\nimage_dims = np.array(image_dims)\naspect_ratios = np.array(aspect_ratios)\n\nprint(f\"Sample image dimensions (width, height): {image_dims[:5]}\")\nprint(f\"Average image size: {image_dims.mean(axis=0)}\")\nprint(f\"Min image size: {image_dims.min(axis=0)}\")\nprint(f\"Max image size: {image_dims.max(axis=0)}\")\nprint(f\"Average aspect ratio: {aspect_ratios.mean():.2f}\")\nprint(f\"Min aspect ratio: {aspect_ratios.min():.2f}\")\nprint(f\"Max aspect ratio: {aspect_ratios.max():.2f}\")\n\n# Check for duplicate or corrupted images\nprint(\"Checking for duplicate or corrupted images...\")\n\nduplicate_count = 0\ncorrupted_count = 0\n\nseen_hashes = set()\nfor _, row in df_train.iterrows():\n    img_path = row['image_path']\n    if os.path.exists(img_path):\n        try:\n            with open(img_path, 'rb') as f:\n                file_hash = hash(f.read())\n            if file_hash in seen_hashes:\n                duplicate_count += 1\n            else:\n                seen_hashes.add(file_hash)\n        except Exception as e:\n            corrupted_count += 1\n    else:\n        corrupted_count += 1\n\nprint(f\"Duplicate images found: {duplicate_count}\")\nprint(f\"Corrupted/missing images found: {corrupted_count}\")\n\nprint(\"Exploratory Data Analysis completed.\")","outputs":[],"cell_number":3,"version":1,"status":"generated","created_at":"2025-11-16T12:33:38.458829+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing\n\nPrepare images and labels for model input, including augmentations and normalization.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_4_START===\")\n\n# ===ALEXANDRIA_CELL_4_START===\nprint(\"===ALEXANDRIA_CELL_4_START===\")\n\n# Preprocessing\n# Prepare images and labels for model input, including augmentations and normalization\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom sklearn.model_selection import train_test_split\nimport albumentations as A\nimport albumentations.pytorch\nimport cv2\nimport json\n\n# Define data path\nDATA_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\nprint(f\"Data path set to: {DATA_PATH}\")\n\n# Load train dataframe from prior cell\n# df_train is assumed to be available from Cell 2\n\n# Split data into train and validation sets (stratified by class)\ntrain_df, val_df = train_test_split(\n    df_train,\n    test_size=0.2,\n    stratify=df_train['label'],\n    random_state=seed\n)\n\nprint(f\"Training set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(val_df)}\")\nprint(f\"Training class distribution:\\n{train_df['label'].value_counts().sort_index()}\")\nprint(f\"Validation class distribution:\\n{val_df['label'].value_counts().sort_index()}\")\n\n# Define image transformations\n# Using albumentations for more flexible augmentations\ntrain_transform = A.Compose([\n    A.Resize(256, 256),\n    A.RandomCrop(224, 224),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n    A.Rotate(limit=15, p=0.2),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n    A.pytorch.ToTensorV2()\n])\n\nval_transform = A.Compose([\n    A.Resize(256, 256),\n    A.CenterCrop(224, 224),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n    A.pytorch.ToTensorV2()\n])\n\n# Custom Dataset class for Cassava Leaf Disease\nclass CassavaLeafDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        # Get image path and label\n        img_path = self.dataframe.iloc[idx]['image_path']\n        label = self.dataframe.iloc[idx]['label']\n        \n        # Load image\n        image = cv2.imread(img_path)\n        if image is None:\n            raise FileNotFoundError(f\"Image not found: {img_path}\")\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        # Apply transformations\n        if self.transform:\n            augmented = self.transform(image=image)\n            image = augmented['image']\n        \n        return image, label\n\n# Create dataset instances\ntrain_dataset = CassavaLeafDataset(train_df, transform=train_transform)\nval_dataset = CassavaLeafDataset(val_df, transform=val_transform)\n\nprint(f\"Train dataset created with {len(train_dataset)} samples.\")\nprint(f\"Validation dataset created with {len(val_dataset)} samples.\")\n\n# Create DataLoader instances\nbatch_size = 32\nnum_workers = 2\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=batch_size,\n    shuffle=True,\n    num_workers=num_workers,\n    pin_memory=True\n)\n\nval_loader = DataLoader(\n    val_dataset,\n    batch_size=batch_size,\n    shuffle=False,\n    num_workers=num_workers,\n    pin_memory=True\n)\n\nprint(f\"Train DataLoader created with batch size {batch_size}.\")\nprint(f\"Validation DataLoader created with batch size {batch_size}.\")\n\n# Verify data loading works\ntry:\n    images, labels = next(iter(train_loader))\n    print(f\"Sample batch - Images shape: {images.shape}, Labels shape: {labels.shape}\")\n    print(\"Data loading successful.\")\nexcept Exception as e:\n    print(f\"Error loading data: {e}\")\n\nprint(\"Preprocessing complete.\")","outputs":[],"cell_number":4,"version":1,"status":"generated","created_at":"2025-11-16T12:33:53.27453+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering\n\nEngineer features or augmentations to improve model robustness and address dataset challenges.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_5_START===\")\n\n# ===ALEXANDRIA_CELL_5_START===\nprint(\"===ALEXANDRIA_CELL_5_START===\")\n\n# Feature Engineering\nprint(\"Starting Feature Engineering...\")\n\nimport torch\nimport torchvision\nimport albumentations as A\nimport albumentations.pytorch\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport random\nimport cv2\nfrom PIL import Image\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom torch.utils.data import WeightedRandomSampler\n\n# Use the same data path as prior cells\nDATA_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\nprint(f\"Data path set to: {DATA_PATH}\")\n\n# Use df_train from prior cell\nprint(f\"Dataset shape: {df_train.shape}\")\n\n# === Advanced Augmentations: CutMix, MixUp, Random Erasing ===\n# Albumentations doesn't have built-in CutMix/MixUp, so we implement them as custom transforms\n# Random Erasing is available in torchvision.transforms\n\nclass CutMix:\n    def __init__(self, alpha=1.0):\n        self.alpha = alpha\n\n    def __call__(self, img1, img2, label1, label2):\n        lam = np.random.beta(self.alpha, self.alpha)\n        h, w = img1.shape[1], img1.shape[2]\n        cx = np.random.randint(0, w)\n        cy = np.random.randint(0, h)\n        cut_w = int(w * np.sqrt(1. - lam))\n        cut_h = int(h * np.sqrt(1. - lam))\n        bbx1 = np.clip(cx - cut_w // 2, 0, w)\n        bby1 = np.clip(cy - cut_h // 2, 0, h)\n        bbx2 = np.clip(cx + cut_w // 2, 0, w)\n        bby2 = np.clip(cy + cut_h // 2, 0, h)\n        img1[:, bby1:bby2, bbx1:bbx2] = img2[:, bby1:bby2, bbx1:bbx2]\n        return img1, lam * label1 + (1 - lam) * label2\n\nclass MixUp:\n    def __init__(self, alpha=1.0):\n        self.alpha = alpha\n\n    def __call__(self, img1, img2, label1, label2):\n        lam = np.random.beta(self.alpha, self.alpha)\n        img = lam * img1 + (1 - lam) * img2\n        label = lam * label1 + (1 - lam) * label2\n        return img, label\n\n# Random Erasing using torchvision\nrandom_erase = torchvision.transforms.RandomErasing(p=0.5, scale=(0.02, 0.33), ratio=(0.3, 3.3), value=0, inplace=False)\n\n# === Handcrafted Features: Color Histograms and Texture (GLCM) ===\ndef extract_color_histogram(image, bins=(8, 8, 8)):\n    hist = cv2.calcHist([image], [0, 1, 2], None, bins, [0, 256, 0, 256, 0, 256])\n    cv2.normalize(hist, hist)\n    return hist.flatten()\n\ndef extract_texture_features(image):\n    gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    # Use a small region for GLCM to reduce computation\n    gray = gray[:128, :128]\n    from skimage.feature import graycomatrix, graycoprops\n    glcm = graycomatrix(gray, distances=[1], angles=[0], levels=256, symmetric=True, normed=True)\n    contrast = graycoprops(glcm, 'contrast')[0, 0]\n    correlation = graycoprops(glcm, 'correlation')[0, 0]\n    energy = graycoprops(glcm, 'energy')[0, 0]\n    homogeneity = graycoprops(glcm, 'homogeneity')[0, 0]\n    return np.array([contrast, correlation, energy, homogeneity])\n\n# Example: Extract features for a few images\nprint(\"Extracting handcrafted features for sample images...\")\nsample_df = df_train.sample(n=5, random_state=seed)\nhandcrafted_features = []\nfor _, row in sample_df.iterrows():\n    img_path = row['image_path']\n    if os.path.exists(img_path):\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        color_hist = extract_color_histogram(image)\n        texture = extract_texture_features(image)\n        features = np.concatenate([color_hist, texture])\n        handcrafted_features.append(features)\n    else:\n        print(f\"Image not found: {img_path}\")\n\nhandcrafted_features = np.array(handcrafted_features)\nprint(f\"Extracted handcrafted features shape: {handcrafted_features.shape}\")\n\n# === Handle Class Imbalance: Weighted Sampling ===\nlabels = df_train['label'].values\nclass_weights = compute_class_weight('balanced', classes=np.unique(labels), y=labels)\nclass_weights = torch.tensor(class_weights, dtype=torch.float)\nsample_weights = [class_weights[label] for label in labels]\nsampler = WeightedRandomSampler(weights=sample_weights, num_samples=len(sample_weights), replacement=True)\n\nprint(f\"Class weights: {class_weights}\")\nprint(f\"Sample weights for first 10 samples: {sample_weights[:10]}\")\n\n# === Visualize Augmented Images ===\n# Use train_transform from prior cell\ndef visualize_augmentations(dataset, num_samples=5):\n    fig, axes = plt.subplots(1, num_samples, figsize=(15, 3))\n    for i in range(num_samples):\n        img, label = dataset[i]\n        # Convert tensor to numpy for visualization\n        img = img.permute(1, 2, 0).numpy()\n        img = np.clip(img, 0, 1)\n        axes[i].imshow(img)\n        axes[i].set_title(f\"Label: {label}\")\n        axes[i].axis('off')\n    plt.tight_layout()\n    plt.show()\n\nprint(\"Visualizing augmented images...\")\nvisualize_augmentations(train_dataset)\n\n# === Summary ===\nprint(\"Feature Engineering completed.\")\nprint(\"Advanced augmentations (CutMix, MixUp, Random Erasing) are ready for use in training loop.\")\nprint(\"Handcrafted features (color histograms, texture) extracted for ensembling.\")\nprint(\"Weighted sampling set up to handle class imbalance.\")","outputs":[],"cell_number":5,"version":1,"status":"generated","created_at":"2025-11-16T12:34:18.453147+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Model Training\n\nTrain a deep learning model (CNN or Vision Transformer) to classify cassava leaf images.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_6_START===\")\n\n# ===ALEXANDRIA_CELL_6_START===\nprint(\"===ALEXANDRIA_CELL_6_START===\")\n\n# Model Training: Train a deep learning model (EfficientNet, ResNet, or ViT) for cassava leaf classification\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nimport time\nimport copy\nimport numpy as np\n\n# Choose model architecture: 'efficientnet', 'resnet', or 'vit'\nMODEL_TYPE = 'efficientnet'  # Change to 'resnet' or 'vit' as desired\n\n# Number of classes (from prior cell)\nnum_classes = df_train['label'].nunique()\nprint(f\"Number of classes: {num_classes}\")\n\n# Device configuration\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# Model selection and initialization\nif MODEL_TYPE == 'efficientnet':\n    print(\"Loading EfficientNet-B0 pretrained model...\")\n    !pip install -q efficientnet_pytorch\n    from efficientnet_pytorch import EfficientNet\n    model = EfficientNet.from_pretrained('efficientnet-b0')\n    in_features = model._fc.in_features\n    model._fc = nn.Linear(in_features, num_classes)\nelif MODEL_TYPE == 'resnet':\n    print(\"Loading ResNet50 pretrained model...\")\n    from torchvision.models import resnet50, ResNet50_Weights\n    model = resnet50(weights=ResNet50_Weights.IMAGENET1K_V2)\n    in_features = model.fc.in_features\n    model.fc = nn.Linear(in_features, num_classes)\nelif MODEL_TYPE == 'vit':\n    print(\"Loading ViT Base Patch16 224 pretrained model...\")\n    !pip install -q timm\n    import timm\n    model = timm.create_model('vit_base_patch16_224', pretrained=True, num_classes=num_classes)\nelse:\n    raise ValueError(f\"Unknown MODEL_TYPE: {MODEL_TYPE}\")\n\nmodel = model.to(device)\nprint(f\"Model {MODEL_TYPE} loaded and moved to {device}.\")\n\n# Loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\nscheduler = ReduceLROnPlateau(optimizer, mode='max', factor=0.5, patience=2, verbose=True)\n\n# Training parameters\nnum_epochs = 10\nbest_acc = 0.0\nbest_model_wts = copy.deepcopy(model.state_dict())\ntrain_losses, val_losses = [], []\ntrain_accuracies, val_accuracies = [], []\n\nprint(\"Starting training loop...\")\n\nfor epoch in range(num_epochs):\n    print(f\"\\nEpoch {epoch+1}/{num_epochs}\")\n    print('-' * 30)\n    since = time.time()\n    \n    # Each epoch has a training and validation phase\n    for phase in ['train', 'val']:\n        if phase == 'train':\n            model.train()\n            dataloader = train_loader\n        else:\n            model.eval()\n            dataloader = val_loader\n        \n        running_loss = 0.0\n        running_corrects = 0\n        total_samples = 0\n        \n        for inputs, labels in dataloader:\n            inputs = inputs.to(device)\n            labels = labels.to(device)\n            \n            optimizer.zero_grad()\n            \n            with torch.set_grad_enabled(phase == 'train'):\n                outputs = model(inputs)\n                loss = criterion(outputs, labels)\n                _, preds = torch.max(outputs, 1)\n                \n                if phase == 'train':\n                    loss.backward()\n                    optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            running_corrects += torch.sum(preds == labels.data)\n            total_samples += inputs.size(0)\n        \n        epoch_loss = running_loss / total_samples\n        epoch_acc = running_corrects.double().item() / total_samples\n        \n        if phase == 'train':\n            train_losses.append(epoch_loss)\n            train_accuracies.append(epoch_acc)\n        else:\n            val_losses.append(epoch_loss)\n            val_accuracies.append(epoch_acc)\n            scheduler.step(epoch_acc)\n        \n        print(f\"{phase.capitalize()} Loss: {epoch_loss:.4f} | {phase.capitalize()} Acc: {epoch_acc:.4f}\")\n        \n        # Deep copy the model if validation accuracy improves\n        if phase == 'val' and epoch_acc > best_acc:\n            best_acc = epoch_acc\n            best_model_wts = copy.deepcopy(model.state_dict())\n            torch.save(best_model_wts, \"best_model.pth\")\n            print(f\"Best model updated and saved at epoch {epoch+1} with val acc {best_acc:.4f}\")\n    \n    time_elapsed = time.time() - since\n    print(f\"Epoch completed in {time_elapsed // 60:.0f}m {time_elapsed % 60:.0f}s\")\n\nprint(\"\\nTraining complete.\")\nprint(f\"Best validation accuracy: {best_acc:.4f}\")\n\n# Load best model weights before inference or submission\nmodel.load_state_dict(best_model_wts)\nprint(\"Best model weights loaded.\")\n\n# Plot training and validation loss/accuracy curves\nimport matplotlib.pyplot as plt\n\nepochs = np.arange(1, num_epochs + 1)\nplt.figure(figsize=(14, 5))\nplt.subplot(1, 2, 1)\nplt.plot(epochs, train_losses, label='Train Loss')\nplt.plot(epochs, val_losses, label='Val Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Loss Curve')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs, train_accuracies, label='Train Acc')\nplt.plot(epochs, val_accuracies, label='Val Acc')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Accuracy Curve')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n\nprint(\"Model training and evaluation curves complete.\")\nprint(\"===ALEXANDRIA_CELL_6_END===\")","outputs":[],"cell_number":6,"version":1,"status":"generated","created_at":"2025-11-16T12:34:29.666289+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Evaluation\n\nEvaluate model performance using accuracy and analyze errors.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_7_START===\")\n\n# ===ALEXANDRIA_CELL_7_START===\nprint(\"===ALEXANDRIA_CELL_7_START===\")\n\n# Evaluation: Compute accuracy, confusion matrix, classification report, visualize misclassifications, cross-validation\n\nimport torch\nimport numpy as np\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\n\n# Use variables from previous cells: model, val_loader, device, val_df, label_mapping\n\nprint(\"Evaluating model on validation set...\")\n\nmodel.eval()\nall_preds = []\nall_labels = []\nall_image_paths = []\n\nwith torch.no_grad():\n    for images, labels in val_loader:\n        images = images.to(device)\n        outputs = model(images)\n        _, preds = torch.max(outputs, 1)\n        all_preds.extend(preds.cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n    # Optionally, collect image paths for misclassification visualization\n    if hasattr(val_loader.dataset, 'dataframe'):\n        all_image_paths = val_loader.dataset.dataframe['image_path'].tolist()\n\nall_preds = np.array(all_preds)\nall_labels = np.array(all_labels)\n\n# Compute overall accuracy\naccuracy = accuracy_score(all_labels, all_preds)\nprint(f\"Validation Accuracy: {accuracy:.4f}\")\n\n# Confusion matrix\ncm = confusion_matrix(all_labels, all_preds)\nnum_classes = len(np.unique(all_labels))\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=[label_mapping[str(i)] for i in range(num_classes)],\n            yticklabels=[label_mapping[str(i)] for i in range(num_classes)])\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()\n\n# Classification report\nreport = classification_report(all_labels, all_preds, target_names=[label_mapping[str(i)] for i in range(num_classes)])\nprint(\"Classification Report:\\n\", report)\n\n# Visualize misclassified images\nprint(\"Visualizing misclassified images...\")\nmis_idx = np.where(all_preds != all_labels)\nnum_to_show = min(12, len(mis_idx))\nif num_to_show > 0:\n    plt.figure(figsize=(15, 6))\n    for i, idx in enumerate(mis_idx[:num_to_show]):\n        img_path = all_image_paths[idx] if len(all_image_paths) > idx else None\n        plt.subplot(2, 6, i + 1)\n        if img_path and os.path.exists(img_path):\n            img = Image.open(img_path)\n            plt.imshow(img)\n        else:\n            plt.text(0.5, 0.5, 'Image not found', ha='center', va='center')\n        plt.axis('off')\n        true_lbl = label_mapping[str(all_labels[idx])]\n        pred_lbl = label_mapping[str(all_preds[idx])]\n        plt.title(f\"T:{true_lbl}\\nP:{pred_lbl}\", fontsize=9)\n    plt.suptitle(\"Sample Misclassified Images (T=True, P=Pred)\", fontsize=14)\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"No misclassified images to display.\")\n\n# Optionally: Cross-validation for robust assessment\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.ensemble import RandomForestClassifier\n\nprint(\"Performing cross-validation with handcrafted features (color histograms + texture)...\")\n\n# Extract handcrafted features for all validation images\ndef extract_features_for_cv(df):\n    features = []\n    for img_path in df['image_path']:\n        if os.path.exists(img_path):\n            image = cv2.imread(img_path)\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            color_hist = cv2.calcHist([image], [0, 1, 2], None, (8, 8, 8), [0, 256, 0, 256, 0, 256])\n            cv2.normalize(color_hist, color_hist)\n            color_hist = color_hist.flatten()\n            gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n            gray = gray[:128, :128]\n            from skimage.feature import graycomatrix, graycoprops\n            glcm = graycomatrix(gray, distances=[1], angles=, levels=256, symmetric=True, normed=True)\n            contrast = graycoprops(glcm, 'contrast')[0, 0]\n            correlation = graycoprops(glcm, 'correlation')[0, 0]\n            energy = graycoprops(glcm, 'energy')[0, 0]\n            homogeneity = graycoprops(glcm, 'homogeneity')[0, 0]\n            texture = np.array([contrast, correlation, energy, homogeneity])\n            feat = np.concatenate([color_hist, texture])\n            features.append(feat)\n        else:\n            features.append(np.zeros(8*8*8 + 4))\n    return np.array(features)\n\nval_features = extract_features_for_cv(val_df)\nval_labels = val_df['label'].values\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nrf = RandomForestClassifier(n_estimators=100, random_state=42, n_jobs=-1)\ncv_scores = cross_val_score(rf, val_features, val_labels, cv=skf, scoring='accuracy')\nprint(f\"Handcrafted feature RF cross-validation accuracy: {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\nprint(\"Evaluation complete.\")\n# ===ALEXANDRIA_CELL_7_END===","outputs":[],"cell_number":7,"version":1,"status":"generated","created_at":"2025-11-16T12:34:41.492969+00:00","metadata":{},"execution_count":null},{"cell_type":"markdown","source":"## Submission/Results\n\nFormat predictions for Kaggle submission and summarize key results.","metadata":{}},{"cell_type":"code","source":"# ⚠️ ALEXANDRIA MARKER - DO NOT DELETE (used for syncing outputs from Kaggle)\nprint(\"===ALEXANDRIA_CELL_8_START===\")\n\n# ===ALEXANDRIA_CELL_8_START===\nprint(\"===ALEXANDRIA_CELL_8_START===\")\n\n# Submission/Results: Format predictions for Kaggle submission and summarize key results\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom tqdm import tqdm\nfrom PIL import Image\nimport cv2\nimport matplotlib.pyplot as plt\n\n# Define data path and test images directory\nDATA_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\ntest_images_dir = os.path.join(DATA_PATH, \"test_images\")\nsample_submission_path = os.path.join(DATA_PATH, \"sample_submission.csv\")\n\n# Check test images directory and sample submission\nprint(f\"Test images directory: {test_images_dir}\")\nprint(f\"Exists: {os.path.exists(test_images_dir)}\")\nprint(f\"Sample submission path: {sample_submission_path}\")\nprint(f\"Exists: {os.path.exists(sample_submission_path)}\")\n\n# Load sample_submission.csv to get test image IDs\nif os.path.exists(sample_submission_path):\n    df_sample_submission = pd.read_csv(sample_submission_path)\n    print(f\"Sample submission loaded. Shape: {df_sample_submission.shape}\")\n    print(df_sample_submission.head())\nelse:\n    raise FileNotFoundError(f\"sample_submission.csv not found at {sample_submission_path}\")\n\n# List all test images\ntest_image_ids = df_sample_submission['image_id'].tolist()\nprint(f\"Number of test images: {len(test_image_ids)}\")\n\n# Prepare test image paths\ntest_image_paths = [os.path.join(test_images_dir, f\"{img_id}.jpg\") for img_id in test_image_ids]\n\n# Use the same validation transform for test images\ntest_transform = val_transform\n\n# Inference: Predict labels for test images\nmodel.eval()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ntest_preds = []\n\nprint(\"Running inference on test images...\")\n\nwith torch.no_grad():\n    for img_path in tqdm(test_image_paths, desc=\"Predicting\"):\n        if not os.path.exists(img_path):\n            print(f\"Warning: Test image not found: {img_path}\")\n            test_preds.append(0)  # fallback to class 0\n            continue\n        # Load and preprocess image\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        augmented = test_transform(image=image)\n        tensor = augmented['image'].unsqueeze(0).to(device)\n        # Predict\n        outputs = model(tensor)\n        _, pred = torch.max(outputs, 1)\n        test_preds.append(int(pred.cpu().item()))\n\n# Prepare submission DataFrame\nsubmission = pd.DataFrame({\n    \"image_id\": test_image_ids,\n    \"label\": test_preds\n})\n\nprint(\"\\nSubmission DataFrame preview:\")\nprint(submission.head())\n\n# Save to CSV\nsubmission_csv_path = \"submission.csv\"\nsubmission.to_csv(submission_csv_path, index=False)\nprint(f\"\\nSubmission file saved to {submission_csv_path}\")\nprint(f\"Submission file shape: {submission.shape}\")\n\n# Display sample predictions (first 10)\nprint(\"\\nSample predictions:\")\ndisplay(submission.head(10))\n\n# Show submission file structure\nprint(\"\\nSubmission file columns:\", submission.columns.tolist())\nprint(\"Unique predicted labels:\", sorted(submission['label'].unique()))\n\n# Visualize a few test predictions\nnum_show = min(8, len(test_image_ids))\nplt.figure(figsize=(16, 4))\nfor i in range(num_show):\n    img_path = test_image_paths[i]\n    if os.path.exists(img_path):\n        img = Image.open(img_path)\n        plt.subplot(1, num_show, i+1)\n        plt.imshow(img)\n        pred_label = submission.iloc[i]['label']\n        pred_name = label_mapping[str(pred_label)] if str(pred_label) in label_mapping else str(pred_label)\n        plt.title(f\"{test_image_ids[i]}\\nPred: {pred_name}\")\n        plt.axis('off')\nplt.suptitle(\"Sample Test Predictions\")\nplt.tight_layout()\nplt.show()\n\n# Summarize final validation accuracy and key findings\nprint(\"\\n=== Final Validation Accuracy ===\")\nif 'accuracy' in locals():\n    print(f\"Validation Accuracy: {accuracy:.4f}\")\nelse:\n    print(\"Validation accuracy variable not found.\")\n\nprint(\"\\n=== Key Findings ===\")\nprint(\"- Model architecture used:\", MODEL_TYPE)\nprint(f\"- Best validation accuracy achieved: {best_acc:.4f}\")\nprint(f\"- Number of classes: {num_classes}\")\nprint(f\"- Submission file ready for upload: {submission_csv_path}\")\nprint(\"- Submission format: [image_id, label]\")\nprint(\"- See above for sample predictions and submission structure.\")\n\nprint(\"Submission/Results cell complete.\")\n# ===ALEXANDRIA_CELL_8_END===","outputs":[],"cell_number":8,"version":1,"status":"generated","created_at":"2025-11-16T12:34:51.900127+00:00","metadata":{},"execution_count":null}],"nbformat":4,"nbformat_minor":4}