{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, MaxPooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.metrics import AUC\nfrom tensorflow.keras.callbacks import Callback\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T04:37:03.735794Z","iopub.execute_input":"2024-11-15T04:37:03.736102Z","iopub.status.idle":"2024-11-15T04:37:16.899954Z","shell.execute_reply.started":"2024-11-15T04:37:03.736068Z","shell.execute_reply":"2024-11-15T04:37:16.898935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Set directories, load csv","metadata":{}},{"cell_type":"code","source":"train_images = '/kaggle/input/histopathologic-cancer-detection/train/'\ntest_images = '/kaggle/input/histopathologic-cancer-detection/test/'\nlabel_csv = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\n\nfull_df = pd.read_csv(label_csv)\n\nfull_df['id'] = full_df['id'] + '.tif'\n\nfull_df['label'] = full_df['label'].astype(str)\n\nprint(full_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:23.230069Z","iopub.execute_input":"2024-11-14T21:52:23.230643Z","iopub.status.idle":"2024-11-14T21:52:23.684707Z","shell.execute_reply.started":"2024-11-14T21:52:23.230576Z","shell.execute_reply":"2024-11-14T21:52:23.682779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:23.686082Z","iopub.execute_input":"2024-11-14T21:52:23.686403Z","iopub.status.idle":"2024-11-14T21:52:23.701502Z","shell.execute_reply.started":"2024-11-14T21:52:23.686368Z","shell.execute_reply":"2024-11-14T21:52:23.700645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Determine label frequency","metadata":{}},{"cell_type":"code","source":"# Calculate frequency distribution\nfrequency_distribution = (full_df.label.value_counts() / len(full_df)).to_frame()\n\n# Plotting the frequency distribution as a bar chart\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']  # light green for benign, light red for malignant\n\n# Plotting bar chart with specified colors\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:23.702463Z","iopub.execute_input":"2024-11-14T21:52:23.702757Z","iopub.status.idle":"2024-11-14T21:52:24.033237Z","shell.execute_reply.started":"2024-11-14T21:52:23.702725Z","shell.execute_reply":"2024-11-14T21:52:24.032327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample images","metadata":{}},{"cell_type":"code","source":"# Sample 16 images and labels from the training set\nsample_images = full_df.sample(16)\n\n# Set up the figure and axes\nfig, axes = plt.subplots(4, 4, figsize=(6, 6))\nfig.tight_layout(pad=1.0)\n\n# Loop through the images and display each one with its label\nfor i, ax in enumerate(axes.flat):\n    # Get the filename and label for each sample\n    id = sample_images.iloc[i]['id']  \n    label = sample_images.iloc[i]['label']  \n\n    # Load the image from file\n    img = mpimg.imread(os.path.join(train_images, id))\n\n    # Display the image\n    ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Label: {label}\")\n    ax.axis('off')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:24.035666Z","iopub.execute_input":"2024-11-14T21:52:24.035971Z","iopub.status.idle":"2024-11-14T21:52:25.428463Z","shell.execute_reply.started":"2024-11-14T21:52:24.035938Z","shell.execute_reply":"2024-11-14T21:52:25.427545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split into train and validation dataframes","metadata":{}},{"cell_type":"code","source":"\n# Split the data into train_df and valid_df with stratified sampling\ntrain_df, valid_df = train_test_split(\n    full_df, \n    test_size=0.2,               # 20% for validation\n    stratify=full_df['label'],     # Stratify by the label column to preserve proportions\n    random_state=42              # Set random seed for reproducibility\n)\n\n# Display the size of each dataset to confirm the split\nprint(f\"Training set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(valid_df)}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:25.429831Z","iopub.execute_input":"2024-11-14T21:52:25.430192Z","iopub.status.idle":"2024-11-14T21:52:25.874552Z","shell.execute_reply.started":"2024-11-14T21:52:25.430152Z","shell.execute_reply":"2024-11-14T21:52:25.873619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensure even label distribution between train and validation images","metadata":{}},{"cell_type":"code","source":"# Calculate the frequencies for training and validation sets\ntrain_frequency = (train_df.label.value_counts() / len(train_df)).to_frame('train_frequency')\nvalid_frequency = (valid_df.label.value_counts() / len(valid_df)).to_frame('valid_frequency')\n\n# Merging the two dataframes to plot them side by side\nfrequency_df = pd.concat([train_frequency, valid_frequency], axis=1)\n\n# Plotting the side-by-side bar chart\nplt.figure(figsize=(8, 5))\nfrequency_df.plot(kind='bar', color=['lightgreen', 'lightcoral'], width=0.8)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels in Train and Validation Sets')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\nplt.legend(['Train Frequency', 'Validation Frequency'], loc='upper right')\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:25.876051Z","iopub.execute_input":"2024-11-14T21:52:25.876341Z","iopub.status.idle":"2024-11-14T21:52:26.189312Z","shell.execute_reply.started":"2024-11-14T21:52:25.876310Z","shell.execute_reply":"2024-11-14T21:52:26.188340Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create data generator","metadata":{}},{"cell_type":"code","source":"# Set batch size\nBATCH_SIZE = 64\n\n# Function to apply random cropping for training images\ndef random_crop(img, crop_size=(32, 32)):\n    # img is expected to be (96, 96, channels)\n    height, width = img.shape[0], img.shape[1]\n    dx, dy = crop_size\n    # Calculate random x, y coordinates for cropping\n    x = np.random.randint(0, height - dx + 1)\n    y = np.random.randint(0, width - dy + 1)\n    return img[x:x+dx, y:y+dy]\n\n# Define data generator with additional augmentations\ntrain_datagen = ImageDataGenerator(\n    rescale=1/255,                 # Normalize pixel values to [0, 1]\n    rotation_range=15,             # Rotate images by up to 15 degrees\n    width_shift_range=0.1,         # Shift images horizontally by 10%\n    height_shift_range=0.1,        # Shift images vertically by 10%\n    horizontal_flip=True,          # Flip images horizontally\n    zoom_range=0.1,                # Zoom in up to 10%\n    brightness_range=(0.8, 1.2)    # Adjust brightness to handle intensity variation\n)\n\n# Data generator for validation, with minimal transformations\nvalid_datagen = ImageDataGenerator(\n    rescale=1/255                  # Only normalization for validation\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:26.190556Z","iopub.execute_input":"2024-11-14T21:52:26.190905Z","iopub.status.idle":"2024-11-14T21:52:26.198365Z","shell.execute_reply.started":"2024-11-14T21:52:26.190870Z","shell.execute_reply":"2024-11-14T21:52:26.197275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create data loaders","metadata":{}},{"cell_type":"code","source":"# Custom data loader to apply random cropping in addition to ImageDataGenerator\nclass CustomDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, generator, dataframe, directory, x_col, y_col, batch_size, target_size, crop_size, shuffle=True):\n        self.generator = generator.flow_from_dataframe(\n            dataframe=dataframe,\n            directory=directory,\n            x_col=x_col,\n            y_col=y_col,\n            class_mode='categorical',\n            batch_size=batch_size,\n            target_size=(96, 96),  # Load original image size\n            shuffle=shuffle\n        )\n        self.crop_size = crop_size\n        self.target_size = target_size\n    \n    def __len__(self):\n        return len(self.generator)\n\n    def __getitem__(self, idx):\n        # Get batch of images and labels\n        batch_x, batch_y = self.generator[idx]\n        # Apply random cropping to each image in the batch\n        batch_x_cropped = np.array([random_crop(img, self.crop_size) for img in batch_x])\n        return batch_x_cropped, batch_y\n\n# Instantiate training and validation loaders\ntrain_loader = CustomDataGenerator(\n    generator=train_datagen,\n    dataframe=train_df,\n    directory=train_images,\n    x_col='id',\n    y_col='label',\n    batch_size=BATCH_SIZE,\n    target_size=(32, 32),\n    crop_size=(32, 32),  # Random crop size\n    shuffle=True\n)\n\nvalid_loader = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=train_images,\n    x_col='id',\n    y_col='label',\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    target_size=(32, 32),\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:52:26.199575Z","iopub.execute_input":"2024-11-14T21:52:26.199900Z","iopub.status.idle":"2024-11-14T21:58:12.665323Z","shell.execute_reply.started":"2024-11-14T21:52:26.199867Z","shell.execute_reply":"2024-11-14T21:58:12.664525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calculate number of steps","metadata":{}},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:58:12.666385Z","iopub.execute_input":"2024-11-14T21:58:12.666670Z","iopub.status.idle":"2024-11-14T21:58:12.671900Z","shell.execute_reply.started":"2024-11-14T21:58:12.666639Z","shell.execute_reply":"2024-11-14T21:58:12.671040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define model","metadata":{}},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, BatchNormalization, MaxPooling2D, Dropout, Dense, Flatten, GlobalAveragePooling2D, LeakyReLU\nfrom tensorflow.keras.regularizers import l2\n\n\ncnn = Sequential([\n    # First convolutional block\n    Conv2D(64, (3, 3), padding='same', input_shape=(32, 32, 3)),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    # Second convolutional block\n    Conv2D(128, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    # Third convolutional block\n    Conv2D(256, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.3),\n\n    # Fourth convolutional block\n    Conv2D(512, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.3),\n\n    # Global Average Pooling instead of Flatten\n    GlobalAveragePooling2D(),\n    \n    # Dense layers with L2 regularization\n    Dense(256, activation='relu', kernel_regularizer=l2(0.001)),\n    BatchNormalization(),\n    Dropout(0.5),\n\n    Dense(128, activation='relu', kernel_regularizer=l2(0.001)),\n    BatchNormalization(),\n    Dropout(0.5),\n\n    # Output layer for binary classification\n    Dense(2, activation='softmax')  # Adjusted for binary classification\n])\n\ncnn.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:58:12.673161Z","iopub.execute_input":"2024-11-14T21:58:12.673495Z","iopub.status.idle":"2024-11-14T21:58:13.689969Z","shell.execute_reply.started":"2024-11-14T21:58:12.673461Z","shell.execute_reply":"2024-11-14T21:58:13.688771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define optimizer and compile model","metadata":{}},{"cell_type":"code","source":"# Define the optimizer\nlearning_rate = 0.0001  # You can adjust this based on your model's performance\noptimizer = Adam(learning_rate=learning_rate)\n\n# Compile the model with the optimizer, a loss function, and metrics\ncnn.compile(optimizer=optimizer, \n            loss='categorical_crossentropy', \n            metrics=[AUC(name='auc')])   # Track accuracy as the performance metric","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:58:13.691369Z","iopub.execute_input":"2024-11-14T21:58:13.691858Z","iopub.status.idle":"2024-11-14T21:58:13.712953Z","shell.execute_reply.started":"2024-11-14T21:58:13.691812Z","shell.execute_reply":"2024-11-14T21:58:13.712218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train model","metadata":{}},{"cell_type":"code","source":"# Define a custom callback to print loss and AUC after each epoch\nclass PrintMetricsCallback(Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        # Access the training and validation loss and AUC from logs\n        train_loss = logs.get('loss')\n        train_auc = logs.get('auc')\n        val_loss = logs.get('val_loss')\n        val_auc = logs.get('val_auc')\n\n        # Print the stats\n        print(f\"Epoch {epoch + 1}/{epochs} - \"\n              f\"Train Loss: {train_loss:.4f}, Train AUC: {train_auc:.4f} - \"\n              f\"Val Loss: {val_loss:.4f}, Val AUC: {val_auc:.4f}\")\n\n# Define number of epochs and an early stopping callback to avoid overfitting\nepochs = 20\nearly_stopping = EarlyStopping(monitor='val_auc', patience=3, restore_best_weights=True)\n\n# Train the model with the custom PrintMetricsCallback\nhistory = cnn.fit(\n    train_loader,\n    validation_data=valid_loader,\n    epochs=epochs,\n    callbacks=[early_stopping, PrintMetricsCallback()],\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T21:58:13.714105Z","iopub.execute_input":"2024-11-14T21:58:13.714461Z","iopub.status.idle":"2024-11-15T01:58:32.072662Z","shell.execute_reply.started":"2024-11-14T21:58:13.714418Z","shell.execute_reply":"2024-11-15T01:58:32.071822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the training and validation accuracy and loss curves\ndef plot_training_curves(history):\n    # Get training and validation metrics\n    auc = history.history['auc']\n    val_auc = history.history['val_auc']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs_range = range(len(auc))\n    \n    plt.figure(figsize=(12, 6))\n\n    # Plot AUC\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, auc, label='Training AUC')\n    plt.plot(epochs_range, val_auc, label='Validation AUC')\n    plt.xlabel('Epochs')\n    plt.ylabel('AUC')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation AUC')\n\n    # Plot Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n\n    plt.show()\n\n# Call the function to display the curves\nplot_training_curves(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:58:32.076540Z","iopub.execute_input":"2024-11-15T01:58:32.076890Z","iopub.status.idle":"2024-11-15T01:58:32.508567Z","shell.execute_reply.started":"2024-11-15T01:58:32.076838Z","shell.execute_reply":"2024-11-15T01:58:32.507635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn.save('/kaggle/working/cnn_wk4.keras')\npickle.dump(history, open(f'cnn_history_v02.pkl', 'wb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T01:58:32.509745Z","iopub.execute_input":"2024-11-15T01:58:32.510041Z","iopub.status.idle":"2024-11-15T01:58:32.815443Z","shell.execute_reply.started":"2024-11-15T01:58:32.510009Z","shell.execute_reply":"2024-11-15T01:58:32.814632Z"}},"outputs":[],"execution_count":null}]}