{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, classification_report\nfrom sklearn.utils import class_weight\n\nfrom itertools import cycle\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import (\n    Layer, Dense, Conv2D, Flatten, MaxPooling2D, Dropout, BatchNormalization,\n    GlobalAveragePooling2D, GlobalMaxPooling2D, Reshape, Multiply, Concatenate, LeakyReLU\n)\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, Callback, ReduceLROnPlateau\nfrom tensorflow.keras.metrics import AUC\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.saving import register_keras_serializable\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:11:58.033711Z","iopub.execute_input":"2024-12-09T22:11:58.034063Z","iopub.status.idle":"2024-12-09T22:12:12.285682Z","shell.execute_reply.started":"2024-12-09T22:11:58.034022Z","shell.execute_reply":"2024-12-09T22:12:12.284651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Set directories, load csv","metadata":{}},{"cell_type":"code","source":"# Paths to data\ntrain_images = '/kaggle/input/histopathologic-cancer-detection/train/'\nlabel_csv = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\n\n# Read labels\nfull_df = pd.read_csv(label_csv)\nfull_df['id'] = full_df['id'] + '.tif'\nfull_df['label'] = full_df['label'].astype(str)\n\n# Filtering function\ndef filter_images(image_dir, image_ids, black_thresh=0.95, white_thresh=0.95):\n    valid_ids = []\n    \n    for img_id in image_ids:\n        img_path = os.path.join(image_dir, img_id)\n        img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        \n        if img is None:\n            continue  # Skip unreadable images\n        \n        # Normalize pixel values to [0, 1]\n        img = img / 255.0\n        \n        # Calculate black and white pixel ratios\n        black_ratio = np.sum(img < 0.1) / img.size\n        white_ratio = np.sum(img > 0.9) / img.size\n        \n        # Keep images that are NOT mostly black or white\n        if black_ratio < black_thresh and white_ratio < white_thresh:\n            valid_ids.append(img_id)\n    \n    return valid_ids\n\n# Filter the images\nvalid_image_ids = filter_images(train_images, full_df['id'].tolist())\nprint(f\"Number of valid images: {len(valid_image_ids)}\")\n\n# Update full_df to include only valid images\nfiltered_df = full_df[full_df['id'].isin(valid_image_ids)].reset_index(drop=True)\nprint(f\"Updated dataframe shape: {filtered_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:12:12.287458Z","iopub.execute_input":"2024-12-09T22:12:12.288007Z","iopub.status.idle":"2024-12-09T22:40:08.239016Z","shell.execute_reply.started":"2024-12-09T22:12:12.287974Z","shell.execute_reply":"2024-12-09T22:40:08.238036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:08.240131Z","iopub.execute_input":"2024-12-09T22:40:08.240408Z","iopub.status.idle":"2024-12-09T22:40:08.256103Z","shell.execute_reply.started":"2024-12-09T22:40:08.240381Z","shell.execute_reply":"2024-12-09T22:40:08.255120Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Determine label frequency","metadata":{}},{"cell_type":"code","source":"# Calculate frequency distribution\nfrequency_distribution = (filtered_df.label.value_counts() / len(filtered_df)).to_frame()\n\n# Plotting the frequency distribution as a bar chart\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']  # light green for benign, light red for malignant\n\n# Plotting bar chart with specified colors\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:08.257240Z","iopub.execute_input":"2024-12-09T22:40:08.257741Z","iopub.status.idle":"2024-12-09T22:40:08.576283Z","shell.execute_reply.started":"2024-12-09T22:40:08.257707Z","shell.execute_reply":"2024-12-09T22:40:08.575462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample images","metadata":{}},{"cell_type":"code","source":"# Sample 16 images and labels from the training set\nsample_images = filtered_df.sample(16)\n\n# Set up the figure and axes\nfig, axes = plt.subplots(4, 4, figsize=(6, 6))\nfig.tight_layout(pad=1.0)\n\n# Loop through the images and display each one with its label\nfor i, ax in enumerate(axes.flat):\n    # Get the filename and label for each sample\n    id = sample_images.iloc[i]['id']  \n    label = sample_images.iloc[i]['label']  \n\n    # Load the image from file\n    img = mpimg.imread(os.path.join(train_images, id))\n\n    # Display the image\n    ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Label: {label}\")\n    ax.axis('off')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:08.579507Z","iopub.execute_input":"2024-12-09T22:40:08.579841Z","iopub.status.idle":"2024-12-09T22:40:09.939203Z","shell.execute_reply.started":"2024-12-09T22:40:08.579815Z","shell.execute_reply":"2024-12-09T22:40:09.938409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split into train and validation dataframes","metadata":{}},{"cell_type":"code","source":"# Split the data into train_df and valid_df with stratified sampling\ntrain_df, valid_df = train_test_split(\n    filtered_df, \n    test_size=0.2,               # 20% for validation\n    stratify=filtered_df['label'],     # Stratify by the label column to preserve proportions\n    random_state=42              # Set random seed for reproducibility\n)\n\n# Display the size of each dataset to confirm the split\nprint(f\"Training set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(valid_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:09.940321Z","iopub.execute_input":"2024-12-09T22:40:09.940640Z","iopub.status.idle":"2024-12-09T22:40:10.200790Z","shell.execute_reply.started":"2024-12-09T22:40:09.940605Z","shell.execute_reply":"2024-12-09T22:40:10.199791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensure no data leakage between train and validation df","metadata":{}},{"cell_type":"code","source":"print(len(set(train_df['id']).intersection(set(valid_df['id']))) == 0)  # Should return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:10.202030Z","iopub.execute_input":"2024-12-09T22:40:10.202738Z","iopub.status.idle":"2024-12-09T22:40:10.277042Z","shell.execute_reply.started":"2024-12-09T22:40:10.202695Z","shell.execute_reply":"2024-12-09T22:40:10.276147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensure even label distribution between train and validation images","metadata":{}},{"cell_type":"code","source":"# Calculate the frequencies for training and validation sets\ntrain_frequency = (train_df.label.value_counts() / len(train_df)).to_frame('train_frequency')\nvalid_frequency = (valid_df.label.value_counts() / len(valid_df)).to_frame('valid_frequency')\n\n# Merging the two dataframes to plot them side by side\nfrequency_df = pd.concat([train_frequency, valid_frequency], axis=1)\n\n# Plotting the side-by-side bar chart\nplt.figure(figsize=(8, 5))\nfrequency_df.plot(kind='bar', color=['lightgreen', 'lightcoral'], width=0.8)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels in Train and Validation Sets')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\nplt.legend(['Train Frequency', 'Validation Frequency'], loc='upper right')\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:10.278285Z","iopub.execute_input":"2024-12-09T22:40:10.278694Z","iopub.status.idle":"2024-12-09T22:40:10.569919Z","shell.execute_reply.started":"2024-12-09T22:40:10.278650Z","shell.execute_reply":"2024-12-09T22:40:10.569156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create data generator and loaders","metadata":{}},{"cell_type":"code","source":"# Set batch size\nBATCH_SIZE = 32\n\n# Combined data generator with augmentations\ntrain_datagen = ImageDataGenerator(\n    rescale=1/255,                # Normalize pixel values to [0, 1]\n    rotation_range=15,            # Rotate images by up to 15 degrees\n    width_shift_range=0.4,        # Horizontal shift (up to 40%)\n    height_shift_range=0.4,       # Vertical shift (up to 40%)\n    horizontal_flip=True,         # Flip images horizontally\n    zoom_range=0.2,               # Zoom in/out slightly\n    fill_mode='reflect'           # Fill mode to handle shifts (options: 'nearest', 'constant', 'reflect', 'wrap')\n)\n\n# Flow the entire training dataframe\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe=train_df,           # Full training dataframe\n    directory=train_images,       # Directory with images\n    x_col='id',                   # Column with image filenames\n    y_col='label',                # Column with labels (0 or 1)\n    class_mode='binary',          # Binary classification\n    target_size=(96, 96),         # Resize images to 96x96\n    batch_size=BATCH_SIZE,        # Batch size\n    shuffle=True                  # Shuffle images during training\n)\n\n# Validation data generator (no augmentations)\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n\nvalid_loader = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,           # Validation dataframe\n    directory=train_images,       # Directory with images\n    x_col='id',\n    y_col='label',\n    class_mode='binary',\n    target_size=(96, 96),\n    batch_size=BATCH_SIZE,\n    shuffle=False                 # No shuffling for validation\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:40:10.571242Z","iopub.execute_input":"2024-12-09T22:40:10.571657Z","iopub.status.idle":"2024-12-09T22:41:50.755440Z","shell.execute_reply.started":"2024-12-09T22:40:10.571613Z","shell.execute_reply":"2024-12-09T22:41:50.754474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(valid_df.head())\nprint(f\"Total validation samples: {len(valid_df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:50.756780Z","iopub.execute_input":"2024-12-09T22:41:50.757162Z","iopub.status.idle":"2024-12-09T22:41:50.764450Z","shell.execute_reply.started":"2024-12-09T22:41:50.757121Z","shell.execute_reply":"2024-12-09T22:41:50.763528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualize Augmentation of Train Images","metadata":{}},{"cell_type":"code","source":"# Fetch a batch of images and labels from the generator\naugmented_images, labels = next(train_loader)\n\n# Plot 9 images from the batch\nplt.figure(figsize=(10, 10))\nfor i in range(9):\n    plt.subplot(3, 3, i+1)\n    plt.imshow(augmented_images[i])  # Display each image\n    plt.axis('off')\n    plt.title(f\"Label: {int(labels[i])}\")  # Show label ID as title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:50.765787Z","iopub.execute_input":"2024-12-09T22:41:50.766141Z","iopub.status.idle":"2024-12-09T22:41:51.573667Z","shell.execute_reply.started":"2024-12-09T22:41:50.766103Z","shell.execute_reply":"2024-12-09T22:41:51.572734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Determine number of steps","metadata":{}},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:51.574843Z","iopub.execute_input":"2024-12-09T22:41:51.575124Z","iopub.status.idle":"2024-12-09T22:41:51.579710Z","shell.execute_reply.started":"2024-12-09T22:41:51.575097Z","shell.execute_reply":"2024-12-09T22:41:51.578895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define CBAM class and define model","metadata":{}},{"cell_type":"code","source":"@register_keras_serializable()\nclass CBAM(Layer):\n    def __init__(self, channels, reduction_ratio=16, **kwargs):\n        super(CBAM, self).__init__(**kwargs)\n        self.channels = channels\n        self.reduction_ratio = reduction_ratio\n\n        # Channel attention layers\n        self.global_avg_pool = GlobalAveragePooling2D()\n        self.global_max_pool = GlobalMaxPooling2D()\n        self.fc1 = Dense(channels // reduction_ratio, activation='relu')\n        self.fc2 = Dense(channels, activation='sigmoid')\n\n        # Spatial attention layers\n        self.conv = Conv2D(1, kernel_size=7, padding='same', activation='sigmoid')\n\n    def build(self, input_shape):\n        # Ensure variables are built once\n        self.reshape_layer = Reshape((1, 1, self.channels))\n        self.concat_layer = Concatenate(axis=-1)\n        self.multiply_layer = Multiply()\n\n    def call(self, inputs):\n        # Channel Attention\n        avg_out = self.global_avg_pool(inputs)\n        max_out = self.global_max_pool(inputs)\n        avg_out = self.fc2(self.fc1(self.reshape_layer(avg_out)))\n        max_out = self.fc2(self.fc1(self.reshape_layer(max_out)))\n        channel_attention = self.multiply_layer([inputs, avg_out + max_out])\n\n        # Spatial Attention\n        avg_pool = tf.reduce_mean(channel_attention, axis=-1, keepdims=True)\n        max_pool = tf.reduce_max(channel_attention, axis=-1, keepdims=True)\n        spatial_attention = self.conv(self.concat_layer([avg_pool, max_pool]))\n        return self.multiply_layer([channel_attention, spatial_attention])\n\n    def get_config(self):\n        config = super(CBAM, self).get_config()\n        config.update({\n            \"channels\": self.channels,\n            \"reduction_ratio\": self.reduction_ratio\n        })\n        return config\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:51.581006Z","iopub.execute_input":"2024-12-09T22:41:51.581250Z","iopub.status.idle":"2024-12-09T22:41:51.594253Z","shell.execute_reply.started":"2024-12-09T22:41:51.581227Z","shell.execute_reply":"2024-12-09T22:41:51.593626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\ndef create_cbam_cnn():\n    cnn = Sequential([\n        # First convolutional block\n        Conv2D(32, (3, 3), padding='same', input_shape=(96, 96, 3)),\n        BatchNormalization(),\n        LeakyReLU(),\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.25),\n\n        # Second convolutional block\n        Conv2D(64, (3, 3), padding='same'),\n        BatchNormalization(),\n        LeakyReLU(),\n        CBAM(64),  # CBAM Block\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.25),\n\n        # Third convolutional block\n        Conv2D(128, (3, 3), padding='same'),\n        BatchNormalization(),\n        LeakyReLU(),\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.3),\n\n        # Fourth convolutional block\n        Conv2D(256, (3, 3), padding='same'),\n        BatchNormalization(),\n        LeakyReLU(),\n        CBAM(256),  # CBAM Block\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.3),\n\n        # Global Average Pooling instead of Flatten\n        GlobalAveragePooling2D(),\n        \n        # Dense layers with L2 regularization\n        Dense(128, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(0.001)),\n        BatchNormalization(),\n        Dropout(0.5),\n\n        Dense(64, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(0.001)),\n        BatchNormalization(),\n        Dropout(0.5),\n\n        # Output layer for binary classification\n        Dense(1, activation='sigmoid')  # Adjusted for binary classification\n    ])\n    return cnn\n\ncnn = create_cbam_cnn()\ncnn.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:51.596828Z","iopub.execute_input":"2024-12-09T22:41:51.597087Z","iopub.status.idle":"2024-12-09T22:41:52.902254Z","shell.execute_reply.started":"2024-12-09T22:41:51.597063Z","shell.execute_reply":"2024-12-09T22:41:52.901413Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define optimizer, assign class weighting and compile model","metadata":{}},{"cell_type":"code","source":"# Define the optimizer\nlearning_rate = 0.0001  # You can adjust this based on your model's performance\noptimizer = Adam(learning_rate=learning_rate)\n\n# Compile the model with the optimizer, a loss function, and metrics\ncnn.compile(optimizer=optimizer, \n            loss='binary_crossentropy', \n            metrics=[AUC(name='auc')])   # Track accuracy as the performance metric\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:52.903265Z","iopub.execute_input":"2024-12-09T22:41:52.903540Z","iopub.status.idle":"2024-12-09T22:41:52.921388Z","shell.execute_reply.started":"2024-12-09T22:41:52.903512Z","shell.execute_reply":"2024-12-09T22:41:52.920808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Determine Class Weights","metadata":{}},{"cell_type":"code","source":"# Get class labels and calculate weights\nclass_labels = train_df['label']  # Replace with actual training labels\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(class_labels),\n    y=class_labels\n)\n\n# Convert to a dictionary\nclass_weights_dict = {i: weight for i, weight in enumerate(class_weights)}\nprint(\"Class Weights:\", class_weights_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:52.922283Z","iopub.execute_input":"2024-12-09T22:41:52.922540Z","iopub.status.idle":"2024-12-09T22:41:53.148268Z","shell.execute_reply.started":"2024-12-09T22:41:52.922510Z","shell.execute_reply":"2024-12-09T22:41:53.147288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train model","metadata":{}},{"cell_type":"code","source":"# Callbacks\nearly_stop = EarlyStopping(monitor='val_auc', patience=5, restore_best_weights=True)\nlr_scheduler = ReduceLROnPlateau(monitor='val_auc', factor=0.5, patience=3, verbose=1)\nmodel_checkpoint = ModelCheckpoint(\"best_cbam_model.keras\", save_best_only=True, monitor=\"val_auc\")\n\n# Specify the number of steps per epoch for the train dataframe\nsteps_per_epoch = len(train_loader)  # Single generator for train dataframe\n\n# Train the model\nhistory = cnn.fit(\n    train_loader,                \n    validation_data=valid_loader,     \n    epochs=30,\n    callbacks=[early_stop, lr_scheduler, model_checkpoint],\n    class_weight=class_weights_dict,\n    verbose=1\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:41:53.149344Z","iopub.execute_input":"2024-12-09T22:41:53.149662Z","iopub.status.idle":"2024-12-10T01:43:38.720806Z","shell.execute_reply.started":"2024-12-09T22:41:53.149633Z","shell.execute_reply":"2024-12-10T01:43:38.719911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Plot results","metadata":{}},{"cell_type":"code","source":"# Plot the training and validation accuracy and loss curves\ndef plot_training_curves(history):\n    # Get training and validation metrics\n    auc = history.history['auc']\n    val_auc = history.history['val_auc']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs_range = range(len(auc))\n    \n    plt.figure(figsize=(12, 6))\n\n    # Plot AUC\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, auc, label='Training AUC')\n    plt.plot(epochs_range, val_auc, label='Validation AUC')\n    plt.xlabel('Epochs')\n    plt.ylabel('AUC')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation AUC')\n\n    # Plot Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n\n    plt.show()\n\n# Call the function to display the curves\nplot_training_curves(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:43:38.722648Z","iopub.execute_input":"2024-12-10T01:43:38.723016Z","iopub.status.idle":"2024-12-10T01:43:39.094644Z","shell.execute_reply.started":"2024-12-10T01:43:38.722978Z","shell.execute_reply":"2024-12-10T01:43:39.093747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Confusion Matrix and Classification Report","metadata":{}},{"cell_type":"code","source":"# Define class names for labels\nclass_names = [\"Benign\", \"Malignant\"]\n\n# Get true labels from validation set\ny_true = valid_loader.labels  # True labels from validation generator\n\n# Predict probabilities for the validation set\ny_pred_prob = cnn.predict(valid_loader)  # Predicted probabilities\n\n# Convert probabilities to binary class predictions\ny_pred = (y_pred_prob > 0.5).astype(int).flatten()  # Binary classification threshold\n\n# Generate the confusion matrix\nconf_matrix = confusion_matrix(y_true, y_pred)\n\n# Display the confusion matrix\nfig, ax = plt.subplots(figsize=(8, 8))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix, display_labels=class_names)\ndisp.plot(cmap=plt.cm.Blues, ax=ax, colorbar=True)\n\nplt.title(\"Confusion Matrix\")\nplt.show()\n\n# Generate the precision-recall report\nprint(\"Classification Report:\")\nreport = classification_report(y_true, y_pred, target_names=class_names)\nprint(report)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:43:39.095776Z","iopub.execute_input":"2024-12-10T01:43:39.096056Z","iopub.status.idle":"2024-12-10T01:44:36.469872Z","shell.execute_reply.started":"2024-12-10T01:43:39.096031Z","shell.execute_reply":"2024-12-10T01:44:36.468901Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Model","metadata":{}},{"cell_type":"code","source":"cnn.save('/kaggle/working/120924v2.keras')\npickle.dump(history, open(f'cnn_history_120724.pkl', 'wb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:44:36.471710Z","iopub.execute_input":"2024-12-10T01:44:36.472093Z","iopub.status.idle":"2024-12-10T01:44:36.697250Z","shell.execute_reply.started":"2024-12-10T01:44:36.472052Z","shell.execute_reply":"2024-12-10T01:44:36.696520Z"}},"outputs":[],"execution_count":null}]}