{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13935636,"sourceType":"datasetVersion","datasetId":8881041}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# SIIM-ISIC Melanoma Classification - EDA Setup\n# Import necessary libraries\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nimport cv2\nfrom pathlib import Path\nimport albumentations as A\nfrom PIL import Image\n\n# Set display options\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', 100)\nsns.set_style('whitegrid')\n\n# Define paths\nbase_path = Path('/kaggle/input/siim-isic-melanoma-classification')\n\n#print(\"Available files in the dataset:\")\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#   for filename in filenames:\n#        print(os.path.join(dirname, filename))\ntrain_csv_path = os.path.join(base_path, 'train.csv')\ntest_csv_path = os.path.join(base_path, 'test.csv')\ntrain_images_dir = os.path.join(base_path, 'jpeg/train/')\ntest_images_dir = os.path.join(base_path, 'jpeg/test/')\n\noutput_dir = Path('/kaggle/working')\nos.makedirs(output_dir / 'eda_plots', exist_ok=True)\nos.makedirs(output_dir / 'augmented', exist_ok=True)","metadata":{"trusted":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2025-12-08T00:14:55.271876Z","iopub.execute_input":"2025-12-08T00:14:55.272448Z","iopub.status.idle":"2025-12-08T00:14:55.279015Z","shell.execute_reply.started":"2025-12-08T00:14:55.272424Z","shell.execute_reply":"2025-12-08T00:14:55.278272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load metadata CSV\ntrain_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n\nprint(\"\\n=== Train DataFrame Overview ===\")\ntrain_df.info()\nprint(f\"Dataset shape: {train_df.shape}\")\nprint(\"\\nFirst 5 rows:\")\nprint(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:01.995981Z","iopub.execute_input":"2025-12-08T00:15:01.996279Z","iopub.status.idle":"2025-12-08T00:15:02.133118Z","shell.execute_reply.started":"2025-12-08T00:15:01.996256Z","shell.execute_reply":"2025-12-08T00:15:02.13244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n=== Initial Data Quality ===\")\nprint(\"Missing values:\\n\", train_df.isnull().sum()[train_df.isnull().sum() > 0])\nprint(f\"Duplicate rows: {train_df.duplicated().sum()}\")\nprint(f\"Duplicate image_names: {train_df['image_name'].duplicated().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:05.432787Z","iopub.execute_input":"2025-12-08T00:15:05.433076Z","iopub.status.idle":"2025-12-08T00:15:05.474346Z","shell.execute_reply.started":"2025-12-08T00:15:05.433053Z","shell.execute_reply":"2025-12-08T00:15:05.47357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n=== Class Distribution ===\")\ncounts = train_df['target'].value_counts().sort_index()\nprint(counts)\nprint(\"-----------------------------------------------------\")\npct = train_df['target'].value_counts(normalize=True) * 100\nprint(pct)\nprint(\"-----------------------------------------------------\")\nratio = counts[0] / counts[1]\n\nprint(f\"Benign: {counts[0]:,} ({pct[0]:.2f}%)\")\nprint(f\"Melanoma: {counts[1]:,} ({pct[1]:.2f}%)\")\nprint(f\"Imbalance ratio: {ratio:.1f}:1\")\n\nwith open(output_dir / 'class_distribution.txt', 'w') as f:\n    f.write(f\"Benign: {counts[0]} ({pct[0]:.2f}%)\\nMelanoma: {counts[1]} ({pct[1]:.2f}%)\\nRatio: {ratio:.1f}:1\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:07.990402Z","iopub.execute_input":"2025-12-08T00:15:07.990996Z","iopub.status.idle":"2025-12-08T00:15:08.004369Z","shell.execute_reply.started":"2025-12-08T00:15:07.99097Z","shell.execute_reply":"2025-12-08T00:15:08.003653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(16, 12))\n# Age\nsns.histplot(train_df['age_approx'], bins=20, kde=True, ax=axes[0,0])\naxes[0,0].set_title('Age Distribution')\n\n# Sex\nsns.countplot(x='sex', data=train_df, ax=axes[0,1])\naxes[0,1].set_title('Sex Distribution')\n\n# Anatomical Site\nsns.countplot(y='anatom_site_general_challenge', data=train_df, ax=axes[1,0])\naxes[1,0].set_title('Anatomical Site Distribution')\n\n# Target (Class Imbalance)\nsns.countplot(x='target', data=train_df, ax=axes[1,1])\naxes[1,1].set_title('Target Distribution (0: Benign, 1: Malignant)')\n\nplt.tight_layout()\nplt.savefig(output_dir / 'eda_plots/metadata.png', dpi=200, bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:11.158803Z","iopub.execute_input":"2025-12-08T00:15:11.159092Z","iopub.status.idle":"2025-12-08T00:15:13.321127Z","shell.execute_reply.started":"2025-12-08T00:15:11.159071Z","shell.execute_reply":"2025-12-08T00:15:13.32026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Diagnosis distribution (train only)\nplt.figure(figsize=(12, 6))\nsns.countplot(y='diagnosis', data=train_df)\nplt.title('Diagnosis Distribution')\nplt.savefig(output_dir / 'eda_plots/diagnosis.png', dpi=200, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:16.737648Z","iopub.execute_input":"2025-12-08T00:15:16.737976Z","iopub.status.idle":"2025-12-08T00:15:17.263845Z","shell.execute_reply.started":"2025-12-08T00:15:16.73795Z","shell.execute_reply":"2025-12-08T00:15:17.263088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n=== Relation entre Variables et Mélanome ===\")\n\nfig, axes = plt.subplots(1, 3, figsize=(20, 5))\n\n# 1. Sexe -> Mélanome\nsex_melanoma = train_df.groupby('sex')['target'].mean() * 100\nsex_melanoma.plot(kind='bar', ax=axes[0], color='red')\naxes[0].set_title('Taux de Mélanome par Sexe')\naxes[0].set_ylabel('Pourcentage Mélanome')\n\n# 2. Âge -> Mélanome\nage_melanoma = train_df.groupby('age_approx')['target'].mean() * 100\nage_melanoma.plot(kind='line', ax=axes[1], color='red', marker='o')\naxes[1].set_title('Taux de Mélanome par Âge')\naxes[1].set_xlabel('Âge')\naxes[1].set_ylabel('Pourcentage Mélanome')\n\n# 3. Site Anatomique -> Mélanome\nsite_melanoma = train_df.groupby('anatom_site_general_challenge')['target'].mean() * 100\nsite_melanoma = site_melanoma.sort_values(ascending=False)  # Trier du plus haut au plus bas\nsite_melanoma.plot(kind='bar', ax=axes[2], color='red')\naxes[2].set_title('Taux de Mélanome par Site Anatomique')\naxes[2].set_ylabel('Pourcentage Mélanome')\naxes[2].tick_params(axis='x', rotation=45)\n\nplt.tight_layout()\nplt.savefig(output_dir / 'eda_plots/melanoma_rates.png', dpi=200, bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:22.194695Z","iopub.execute_input":"2025-12-08T00:15:22.195259Z","iopub.status.idle":"2025-12-08T00:15:23.555837Z","shell.execute_reply.started":"2025-12-08T00:15:22.195238Z","shell.execute_reply":"2025-12-08T00:15:23.555108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop NA in 'sex' (as suggested: only 65 rows, and unknown is OK to drop)\ntrain_df = train_df.dropna(subset=['sex'])\n# Fill missing values without inplace parameter to avoid FutureWarning\ntrain_df['age_approx'] = train_df['age_approx'].fillna(train_df['age_approx'].mean())\ntrain_df['anatom_site_general_challenge'] = train_df['anatom_site_general_challenge'].fillna('unknown')\n\n\nprint(\"\\nAfter cleaning:\")\nprint(f\"Total images: {len(train_df):,}\")\nprint(f\"Missing values: {train_df.isnull().sum().sum()}\")\nprint(f\"Duplicate rows: {train_df.duplicated().sum()}\")\nprint(f\"Duplicate image_names: {train_df['image_name'].duplicated().sum()}\")\nprint(f\"Malignant cases remaining: {train_df['target'].sum():,}\")\nprint(f\"After cleaning: {len(train_df):,} images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:26.482668Z","iopub.execute_input":"2025-12-08T00:15:26.483239Z","iopub.status.idle":"2025-12-08T00:15:26.522486Z","shell.execute_reply.started":"2025-12-08T00:15:26.483216Z","shell.execute_reply":"2025-12-08T00:15:26.521741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Patient-level analysis\npatient_counts = train_df['patient_id'].value_counts()\nmelanoma_patients = train_df[train_df['target']==1]['patient_id'].nunique()\n\nprint(f\"\\nUnique patients: {train_df['patient_id'].nunique():,}\")\nprint(f\"Mean images/patient: {patient_counts.mean():.1f} (max: {patient_counts.max()})\")\nprint(f\"Patients with melanoma: {melanoma_patients:,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:29.432702Z","iopub.execute_input":"2025-12-08T00:15:29.433287Z","iopub.status.idle":"2025-12-08T00:15:29.44415Z","shell.execute_reply.started":"2025-12-08T00:15:29.433263Z","shell.execute_reply":"2025-12-08T00:15:29.443408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Sample Image Visualization (load a few to avoid memory issues)\ndef load_and_show_images(df, target_value, num_samples=5):\n    samples = df[df['target'] == target_value].sample(num_samples)\n    plt.figure(figsize=(15, 5))\n    for i, (idx, row) in enumerate(samples.iterrows()):\n        img_path = os.path.join(train_images_dir, row['image_name'] + '.jpg')\n        if os.path.exists(img_path):\n            img = cv2.imread(img_path)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            plt.subplot(1, num_samples, i+1)\n            plt.imshow(img)\n            plt.title(f\"Target: {target_value}\")\n            plt.axis('off')\n    plt.show()\nprint(\"\\nSample Benign Images:\")\nload_and_show_images(train_df, 0)\nprint(\"\\nSample Malignant Images:\")\nload_and_show_images(train_df, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:31.874636Z","iopub.execute_input":"2025-12-08T00:15:31.875306Z","iopub.status.idle":"2025-12-08T00:15:39.770531Z","shell.execute_reply.started":"2025-12-08T00:15:31.875281Z","shell.execute_reply":"2025-12-08T00:15:39.769783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Improved Pixel & Image Size Stats (Sample 100 images)\nsample_df = train_df.sample(100, random_state=42)\npixel_means = []\nimage_sizes = []\n\nfor _, row in sample_df.iterrows():\n    img_path = os.path.join(train_images_dir, row['image_name'] + '.jpg')\n    if os.path.exists(img_path):\n        img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        pixel_means.append(img.mean())\n        image_sizes.append(img.shape)  # Get (height, width)\n\n# Convert to arrays for easier analysis\nheights = [size[0] for size in image_sizes]\nwidths = [size[1] for size in image_sizes]\n\nprint(\"\\n=== Image Statistics (Sampled) ===\")\nprint(f\"Mean Pixel Intensity: {np.mean(pixel_means):.2f} ± {np.std(pixel_means):.2f}\")\nprint(f\"Image Heights: {np.mean(heights):.1f} ± {np.std(heights):.1f}\")\nprint(f\"Image Widths: {np.mean(widths):.1f} ± {np.std(widths):.1f}\")\nprint(f\"Min Size: {min(heights)}x{min(widths)}\")\nprint(f\"Max Size: {max(heights)}x{max(widths)}\")\n\n# Check if all images are same size\nunique_sizes = set(image_sizes)\nprint(f\"Unique image sizes: {len(unique_sizes)}\")\nif len(unique_sizes) > 1:\n    print(\"⚠️  Images have different sizes - will need resizing for training!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:42.46962Z","iopub.execute_input":"2025-12-08T00:15:42.470462Z","iopub.status.idle":"2025-12-08T00:15:47.916749Z","shell.execute_reply.started":"2025-12-08T00:15:42.470437Z","shell.execute_reply":"2025-12-08T00:15:47.916075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['sex', 'anatom_site_general_challenge']\nencoded = pd.get_dummies(train_df[cat_cols], drop_first=True)\ncorr_df = pd.concat([train_df[['age_approx']], encoded], axis=1)\n\n# 1. Feature-to-feature correlation\nplt.figure(figsize=(12, 9))\nsns.heatmap(corr_df.corr(), annot=False, cmap='coolwarm', center=0, linewidths=0.5)\nplt.title('Feature Correlation (Excluding Target)')\nplt.savefig(output_dir / 'eda_plots/feature_corr.png', dpi=200, bbox_inches='tight')\nplt.show()\n\n# 2. Feature-to-target correlation\ntarget_corr = corr_df.corrwith(train_df['target']).sort_values(ascending=False)\n\nplt.figure(figsize=(6, 8))\nsns.heatmap(target_corr.to_frame('Correlation with Target'), \n            annot=True, cmap='coolwarm', fmt='.3f')\nplt.title('Feature Importance via Correlation with Melanoma')\nplt.savefig(output_dir / 'eda_plots/target_corr.png', dpi=200, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:50.904797Z","iopub.execute_input":"2025-12-08T00:15:50.90532Z","iopub.status.idle":"2025-12-08T00:15:52.346984Z","shell.execute_reply.started":"2025-12-08T00:15:50.905297Z","shell.execute_reply":"2025-12-08T00:15:52.346236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Patient-Wise Stratified Split\ntrain_images_dir = Path('/kaggle/input/siim-isic-melanoma-classification/jpeg/train/')\npatient_target = (\n    train_df.groupby('patient_id')['target']\n    .max()\n    .reset_index()\n    .rename(columns={'target': 'has_melanoma'})\n)\n\np_train, p_temp = train_test_split(\n    patient_target, test_size=0.30, stratify=patient_target['has_melanoma'], random_state=42\n)\np_val, p_test = train_test_split(\n    p_temp, test_size=0.50, stratify=p_temp['has_melanoma'], random_state=42\n)\n\ntrain_split = train_df[train_df['patient_id'].isin(p_train['patient_id'])].copy()\nval_split   = train_df[train_df['patient_id'].isin(p_val['patient_id'])].copy()\ntest_split  = train_df[train_df['patient_id'].isin(p_test['patient_id'])].copy()\n\n# --- Add image_path using Path (NOW WORKS) ---\nfor df in [train_split, val_split, test_split]:\n    df['image_path'] = df['image_name'].apply(\n        lambda x: str(train_images_dir / f\"{x}.jpg\")   # Path + str → Path → str\n    )\n\n# --- Save ---\ntrain_split.to_csv(output_dir / 'train_split.csv', index=False)\nval_split.to_csv(output_dir / 'val_split.csv', index=False)\ntest_split.to_csv(output_dir / 'test_split.csv', index=False)\n\nprint(f\"Train: {len(train_split):,} | Val: {len(val_split):,} | Test: {len(test_split):,}\")\nprint(f\"Melanoma rate → Train: {train_split['target'].mean():.4f}, Val: {val_split['target'].mean():.4f}, Test: {test_split['target'].mean():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:56.19448Z","iopub.execute_input":"2025-12-08T00:15:56.195162Z","iopub.status.idle":"2025-12-08T00:15:56.517111Z","shell.execute_reply.started":"2025-12-08T00:15:56.195136Z","shell.execute_reply":"2025-12-08T00:15:56.516343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport cv2\n\n# Simpler but effective augmentation pipeline\ntransform = A.Compose([\n    A.Resize(224, 224),  # Simple resize instead of RandomResizedCrop\n    A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.2, rotate_limit=90, p=0.5),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.5),\n    A.HueSaturationValue(hue_shift_limit=10, sat_shift_limit=20, val_shift_limit=10, p=0.5),\n    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n    ToTensorV2()\n])\n\nprint(\"✅ Advanced augmentation pipeline loaded!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:15:59.420652Z","iopub.execute_input":"2025-12-08T00:15:59.421377Z","iopub.status.idle":"2025-12-08T00:15:59.436841Z","shell.execute_reply.started":"2025-12-08T00:15:59.421351Z","shell.execute_reply":"2025-12-08T00:15:59.435631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get only melanoma images from your train_split\nmelanoma = train_split[train_split['target'] == 1]\naug_records = []\nk = 3  # Create 3 new versions of each melanoma image\n\nprint(f\"Starting advanced augmentation for {len(melanoma)} melanoma images...\")\n\nfor idx, row in melanoma.iterrows():\n    # Load image with OpenCV (required for albumentations)\n    img = cv2.imread(row['image_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    for i in range(k):\n        # Apply your advanced augmentation\n        augmented = transform(image=img)\n        aug_img = augmented['image']  # This is now a tensor\n        \n        # Convert tensor back to numpy for saving\n        aug_img_np = aug_img.permute(1, 2, 0).numpy()  # Convert from (C,H,W) to (H,W,C)\n        aug_img_np = (aug_img_np * 255).astype(np.uint8)  # Denormalize to 0-255\n        \n        # Create new filename and path\n        name = f\"{row['image_name']}_aug_{i}.jpg\"\n        path = output_dir / 'augmented' / name\n        \n        # Save the new image\n        Image.fromarray(aug_img_np).save(path)\n        \n        # Create new database record\n        new_row = row.copy()\n        new_row['image_name'] = name[:-4]  # Remove .jpg\n        new_row['image_path'] = str(path)\n        aug_records.append(new_row)\n    \n    if idx % 50 == 0:\n        print(f\"Augmented {idx+1}/{len(melanoma)} melanoma images...\")\n\n# Save augmented data to CSV\naug_df = pd.DataFrame(aug_records)\naug_df.to_csv(output_dir / 'augmented_melanoma.csv', index=False)\nprint(f\"✅ Created {len(aug_df):,} new melanoma images with advanced augmentation!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:04.075882Z","iopub.execute_input":"2025-12-08T00:16:04.076547Z","iopub.status.idle":"2025-12-08T00:16:26.742498Z","shell.execute_reply.started":"2025-12-08T00:16:04.076525Z","shell.execute_reply":"2025-12-08T00:16:26.741864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This combines original data + new augmented melanoma images\nprint(\"=== Combining Original + Augmented Data ===\")\n\n# Combine original train_split with new augmented melanoma\nbalanced_train_df = pd.concat([train_split, aug_df], ignore_index=True)\n\nprint(f\"Original training size: {len(train_split):,} images\")\nprint(f\"Augmented melanoma: {len(aug_df):,} images\") \nprint(f\"Balanced dataset: {len(balanced_train_df):,} images\")\nprint(f\"Melanoma ratio improved: {train_split['target'].mean():.3f} → {balanced_train_df['target'].mean():.3f}\")\n\n# Save the balanced dataset\nbalanced_train_df.to_csv(output_dir / 'balanced_train.csv', index=False)\nprint(\"✅ Saved balanced_train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:29.258651Z","iopub.execute_input":"2025-12-08T00:16:29.259347Z","iopub.status.idle":"2025-12-08T00:16:29.394187Z","shell.execute_reply.started":"2025-12-08T00:16:29.259323Z","shell.execute_reply":"2025-12-08T00:16:29.393505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This shows you what was created\nprint(\"=== Data Verification ===\")\nprint(\"Original train_split shape:\", train_split.shape)\nprint(\"Augmented data shape:\", aug_df.shape)\nprint(\"Balanced data shape:\", balanced_train_df.shape)\n\nprint(\"\\nClass distribution:\")\nprint(\"Before augmentation:\")\nprint(train_split['target'].value_counts())\nprint(\"\\nAfter augmentation:\")\nprint(balanced_train_df['target'].value_counts())\n\nprint(\"\\nSample of new augmented images:\")\nprint(aug_df[['image_name', 'sex', 'age_approx', 'anatom_site_general_challenge']].head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:31.842734Z","iopub.execute_input":"2025-12-08T00:16:31.843015Z","iopub.status.idle":"2025-12-08T00:16:31.852557Z","shell.execute_reply.started":"2025-12-08T00:16:31.842994Z","shell.execute_reply":"2025-12-08T00:16:31.851788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Model\nimport os\n\nprint(\"🚀 Setting up ResNet50...\")\n\nweights_path = \"/kaggle/input/weights/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\"\n\nbase_model = ResNet50(\n    weights=weights_path,\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\nprint(\"✅ Pre-trained ResNet50 loaded from local file!\")\n\n# Build custom classification head\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(128, activation='relu')(x)\nx = Dropout(0.3)(x)\npredictions = Dense(1, activation='sigmoid')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\nbase_model.trainable = False\n\nprint(\"✅ Model architecture built!\")\nprint(f\"Input shape: {model.input_shape}\")\nprint(f\"Output shape: {model.output_shape}\")\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:34.698811Z","iopub.execute_input":"2025-12-08T00:16:34.69937Z","iopub.status.idle":"2025-12-08T00:16:39.009676Z","shell.execute_reply.started":"2025-12-08T00:16:34.699346Z","shell.execute_reply":"2025-12-08T00:16:39.008874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\n\nprint(\"⚙️ Compiling model...\")\n\nmodel.compile(\n    optimizer=Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['accuracy', 'AUC', 'Precision', 'Recall']\n)\n\nprint(\"✅ Model compiled!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:42.72424Z","iopub.execute_input":"2025-12-08T00:16:42.724509Z","iopub.status.idle":"2025-12-08T00:16:42.743322Z","shell.execute_reply.started":"2025-12-08T00:16:42.72449Z","shell.execute_reply":"2025-12-08T00:16:42.742608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nprint(\"📊 Creating data generators...\")\n\n# Simple generators - only normalization\ntrain_datagen = ImageDataGenerator(rescale=1./255)\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    balanced_train_df,\n    x_col='image_path',\n    y_col='target',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='raw',      # This gives targets as (32,) shape\n    shuffle=True\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    val_split,\n    x_col='image_path', \n    y_col='target',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='raw',      # Same here\n    shuffle=False\n)\n\nprint(f\"✅ Training samples: {len(balanced_train_df):,}\")\nprint(f\"✅ Validation samples: {len(val_split):,}\")\n\n# Verify shapes\nsample_batch = next(train_generator)\nprint(f\"Batch image shape: {sample_batch[0].shape}\")  # Should be (32, 224, 224, 3)\nprint(f\"Batch target shape: {sample_batch[1].shape}\") # Should be (32,)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:16:44.832623Z","iopub.execute_input":"2025-12-08T00:16:44.832916Z","iopub.status.idle":"2025-12-08T00:17:33.38344Z","shell.execute_reply.started":"2025-12-08T00:16:44.832894Z","shell.execute_reply":"2025-12-08T00:17:33.382756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\n\nprint(\"⚖️ Calculating class weights...\")\n\nclass_weights = compute_class_weight(\n    'balanced',\n    classes=[0, 1],\n    y=balanced_train_df['target']\n)\nclass_weight_dict = {0: class_weights[0], 1: class_weights[1]}  \nprint(f\"Class weights: {class_weight_dict}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T00:17:33.69033Z","iopub.execute_input":"2025-12-08T00:17:33.690893Z","iopub.status.idle":"2025-12-08T00:17:33.699455Z","shell.execute_reply.started":"2025-12-08T00:17:33.690871Z","shell.execute_reply":"2025-12-08T00:17:33.698765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\n\nprint(\"🎯 Starting training...\")\n\n# FIXED: Added mode='max' for AUC\ncallbacks = [\n    ModelCheckpoint(\n        output_dir / 'pretrained_resnet50_melanoma.h5',\n        monitor='val_auc',\n        save_best_only=True,\n        mode='max', \n        verbose=1\n    ),\n    EarlyStopping(\n        monitor='val_auc',\n        patience=5,\n        restore_best_weights=True,\n        mode='max', \n        verbose=1\n    )\n]\n\n# Train the model\nhistory = model.fit(\n    train_generator,\n    epochs=15,\n    validation_data=val_generator,\n    class_weight=class_weight_dict,\n    callbacks=callbacks,\n    verbose=1\n)\n\nprint(\"✅ Training completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-08T10:35:07.411605Z","iopub.execute_input":"2025-12-08T10:35:07.411845Z","iopub.status.idle":"2025-12-08T10:35:20.815864Z","shell.execute_reply.started":"2025-12-08T10:35:07.411821Z","shell.execute_reply":"2025-12-08T10:35:20.814692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, roc_auc_score, roc_curve\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\nprint(\"📊 Evaluating the model...\")\n\n# Load the best saved model\nbest_model = tf.keras.models.load_model(output_dir / 'pretrained_resnet50_melanoma.h5')\nprint(\"✅ Best model loaded!\")\n\n# 1. Make predictions on validation data\nprint(\"🔍 Making predictions...\")\nval_generator.reset()  # Reset generator to start from beginning\n\n# Get all predictions\ny_pred_proba = best_model.predict(val_generator, verbose=1)\ny_pred = (y_pred_proba > 0.5).astype(int).flatten()\n\n# Get true labels\ny_true = val_generator.labels\n\nprint(f\"\\n📈 Predictions shape: {y_pred_proba.shape}\")\nprint(f\"True labels shape: {y_true.shape}\")\n\n# 2. Calculate evaluation metrics\nprint(\"\\n\" + \"=\"*50)\nprint(\"MODEL EVALUATION RESULTS\")\nprint(\"=\"*50)\n\n# ROC-AUC\nauc_score = roc_auc_score(y_true, y_pred_proba)\nprint(f\"\\n📊 ROC-AUC Score: {auc_score:.4f}\")\n\n# Classification report\nprint(\"\\n📋 Classification Report:\")\nprint(classification_report(y_true, y_pred, target_names=['Benign', 'Malignant']))\n\n# Confusion matrix\ncm = confusion_matrix(y_true, y_pred)\nprint(\"\\n🎯 Confusion Matrix:\")\nprint(cm)\n\n# 3. Additional detailed metrics\ntn, fp, fn, tp = cm.ravel()\n\nsensitivity = tp / (tp + fn) if (tp + fn) > 0 else 0\nspecificity = tn / (tn + fp) if (tn + fp) > 0 else 0\nprecision = tp / (tp + fp) if (tp + fp) > 0 else 0\nf1_score = 2 * (precision * sensitivity) / (precision + sensitivity) if (precision + sensitivity) > 0 else 0\n\nprint(f\"\\n📊 Detailed Metrics:\")\nprint(f\"Sensitivity (Recall): {sensitivity:.4f}\")\nprint(f\"Specificity: {specificity:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"F1-Score: {f1_score:.4f}\")\nprint(f\"Accuracy: {(tp + tn) / (tp + tn + fp + fn):.4f}\")\n\n# 4. Plot ROC Curve\nprint(\"\\n📈 Generating ROC Curve...\")\nfpr, tpr, thresholds = roc_curve(y_true, y_pred_proba)\n\nplt.figure(figsize=(10, 8))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (AUC = {auc_score:.3f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--', label='Random')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc=\"lower right\")\nplt.grid(True, alpha=0.3)\n\n# Save ROC curve\nroc_path = output_dir / 'roc_curve.png'\nplt.savefig(roc_path, dpi=300, bbox_inches='tight')\nplt.close()\nprint(f\"✅ ROC curve saved to: {roc_path}\")\n\n# 5. Plot Confusion Matrix Heatmap\nprint(\"\\n📊 Generating Confusion Matrix Heatmap...\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n            xticklabels=['Predicted Benign', 'Predicted Malignant'],\n            yticklabels=['Actual Benign', 'Actual Malignant'])\nplt.title('Confusion Matrix Heatmap')\nplt.ylabel('Actual')\nplt.xlabel('Predicted')\n\n# Save confusion matrix\ncm_path = output_dir / 'confusion_matrix.png'\nplt.savefig(cm_path, dpi=300, bbox_inches='tight')\nplt.close()\nprint(f\"✅ Confusion matrix saved to: {cm_path}\")\n\n# 6. Plot Precision-Recall Curve\nprint(\"\\n📈 Generating Precision-Recall Curve...\")\nfrom sklearn.metrics import precision_recall_curve, average_precision_score\n\nprecision_vals, recall_vals, _ = precision_recall_curve(y_true, y_pred_proba)\navg_precision = average_precision_score(y_true, y_pred_proba)\n\nplt.figure(figsize=(10, 8))\nplt.plot(recall_vals, precision_vals, color='darkgreen', lw=2, \n         label=f'Precision-Recall curve (AP = {avg_precision:.3f})')\nplt.xlabel('Recall')\nplt.ylabel('Precision')\nplt.title('Precision-Recall Curve')\nplt.legend(loc=\"upper right\")\nplt.grid(True, alpha=0.3)\n\n# Save Precision-Recall curve\npr_path = output_dir / 'precision_recall_curve.png'\nplt.savefig(pr_path, dpi=300, bbox_inches='tight')\nplt.close()\nprint(f\"✅ Precision-Recall curve saved to: {pr_path}\")\n\n# 7. Distribution of predictions\nprint(\"\\n📊 Analyzing prediction distribution...\")\nplt.figure(figsize=(10, 6))\n\n# Histogram of predicted probabilities\nplt.hist(y_pred_proba[y_true == 0], bins=50, alpha=0.5, label='Benign', color='blue')\nplt.hist(y_pred_proba[y_true == 1], bins=50, alpha=0.5, label='Malignant', color='red')\nplt.xlabel('Predicted Probability (Malignant)')\nplt.ylabel('Frequency')\nplt.title('Distribution of Predicted Probabilities')\nplt.legend()\nplt.grid(True, alpha=0.3)\n\n# Save distribution plot\ndist_path = output_dir / 'prediction_distribution.png'\nplt.savefig(dist_path, dpi=300, bbox_inches='tight')\nplt.close()\nprint(f\"✅ Prediction distribution saved to: {dist_path}\")\n\n# 8. Find optimal threshold\nprint(\"\\n🔍 Finding optimal threshold...\")\n# Youden's J statistic\nyouden_j = tpr - fpr\noptimal_idx = np.argmax(youden_j)\noptimal_threshold = thresholds[optimal_idx]\n\nprint(f\"Optimal threshold (Youden's J): {optimal_threshold:.3f}\")\nprint(f\"At this threshold:\")\nprint(f\"  - Sensitivity: {tpr[optimal_idx]:.3f}\")\nprint(f\"  - Specificity: {1 - fpr[optimal_idx]:.3f}\")\n\n# 9. Evaluation with optimal threshold\nprint(\"\\n\" + \"=\"*50)\nprint(\"EVALUATION WITH OPTIMAL THRESHOLD\")\nprint(\"=\"*50)\n\ny_pred_optimal = (y_pred_proba > optimal_threshold).astype(int).flatten()\ncm_optimal = confusion_matrix(y_true, y_pred_optimal)\n\nprint(\"\\n🎯 Confusion Matrix (Optimal Threshold):\")\nprint(cm_optimal)\n\ntn_opt, fp_opt, fn_opt, tp_opt = cm_optimal.ravel()\n\nprint(f\"\\n📊 Metrics at optimal threshold ({optimal_threshold:.3f}):\")\nprint(f\"Accuracy: {(tp_opt + tn_opt) / len(y_true):.4f}\")\nprint(f\"Sensitivity: {tp_opt / (tp_opt + fn_opt):.4f}\")\nprint(f\"Specificity: {tn_opt / (tn_opt + fp_opt):.4f}\")\nprint(f\"Precision: {tp_opt / (tp_opt + fp_opt) if (tp_opt + fp_opt) > 0 else 0:.4f}\")\n\nprint(\"\\n✅ Evaluation completed!\")\nprint(f\"📁 All plots saved in: {output_dir}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T21:04:28.763858Z","iopub.execute_input":"2025-12-07T21:04:28.764248Z","iopub.status.idle":"2025-12-07T21:04:30.982558Z","shell.execute_reply.started":"2025-12-07T21:04:28.764223Z","shell.execute_reply":"2025-12-07T21:04:30.981489Z"}},"outputs":[],"execution_count":null}]}