{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":194046,"sourceType":"modelInstanceVersion","modelInstanceId":165482,"modelId":187814}],"dockerImageVersionId":30788,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2\nimport tensorflow as tf\nimport random\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, classification_report\nfrom sklearn.utils import class_weight\n\nfrom itertools import cycle\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import (\n    Layer, Dense, Conv2D, Flatten, MaxPooling2D, Dropout, BatchNormalization,\n    GlobalAveragePooling2D, GlobalMaxPooling2D, Reshape, Multiply, Concatenate, LeakyReLU\n)\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, Callback, ReduceLROnPlateau\nfrom tensorflow.keras.metrics import AUC\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.saving import register_keras_serializable\nfrom tensorflow import keras\n\n\n# Set global seeds for reproducibility\nSEED = 42\ntf.random.set_seed(SEED)\nnp.random.seed(SEED)\nrandom.seed(SEED)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:01.592868Z","iopub.execute_input":"2024-12-09T21:09:01.593252Z","iopub.status.idle":"2024-12-09T21:09:01.618948Z","shell.execute_reply.started":"2024-12-09T21:09:01.593217Z","shell.execute_reply":"2024-12-09T21:09:01.617640Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"test_images = '/kaggle/input/histopathologic-cancer-detection/test/'\n\ntest_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\ntest_df['id'] = test_df['id'] + '.tif'\ntest_df['label'] = test_df['label'].astype(str)\n\nprint('Test Set Size:', test_df.shape)\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:01.621096Z","iopub.execute_input":"2024-12-09T21:09:01.621520Z","iopub.status.idle":"2024-12-09T21:09:01.727959Z","shell.execute_reply.started":"2024-12-09T21:09:01.621482Z","shell.execute_reply":"2024-12-09T21:09:01.726791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# View Image Sample","metadata":{}},{"cell_type":"code","source":"# Sample 16 images and labels from the training set\nsample_images = test_df.sample(16)\n\n# Set up the figure and axes\nfig, axes = plt.subplots(4, 4, figsize=(6, 6))\nfig.tight_layout(pad=1.0)\n\n# Loop through the images and display each one with its label\nfor i, ax in enumerate(axes.flat):\n    # Get the filename and label for each sample\n    id = sample_images.iloc[i]['id']  \n    label = sample_images.iloc[i]['label']  \n\n    # Load the image from file\n    img = mpimg.imread(os.path.join(test_images, id))\n\n    # Display the image\n    ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Label: {label}\")\n    ax.axis('off')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:01.729738Z","iopub.execute_input":"2024-12-09T21:09:01.730182Z","iopub.status.idle":"2024-12-09T21:09:02.977051Z","shell.execute_reply.started":"2024-12-09T21:09:01.730144Z","shell.execute_reply":"2024-12-09T21:09:02.975746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create data generator and loader","metadata":{}},{"cell_type":"code","source":"# Data generator\ntest_datagen = ImageDataGenerator(\n    rescale=1/255,\n) # Normalize pixel values\n\n# Data loader\ntest_loader = test_datagen.flow_from_dataframe(\n    dataframe = test_df,\n    directory = test_images,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = 32,\n    seed = 1,\n    shuffle = False,\n    class_mode = None,\n    target_size = (96,96)\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:02.982158Z","iopub.execute_input":"2024-12-09T21:09:02.982713Z","iopub.status.idle":"2024-12-09T21:09:38.896527Z","shell.execute_reply.started":"2024-12-09T21:09:02.982661Z","shell.execute_reply":"2024-12-09T21:09:38.895257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create CBAM object","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import register_keras_serializable\n\n@register_keras_serializable(package=\"Custom\")\n@register_keras_serializable()\nclass CBAM(Layer):\n    def __init__(self, channels, reduction_ratio=16, **kwargs):\n        super(CBAM, self).__init__(**kwargs)\n        self.channels = channels\n        self.reduction_ratio = reduction_ratio\n\n        # Channel attention layers\n        self.global_avg_pool = GlobalAveragePooling2D()\n        self.global_max_pool = GlobalMaxPooling2D()\n        self.fc1 = Dense(channels // reduction_ratio, activation='relu')\n        self.fc2 = Dense(channels, activation='sigmoid')\n\n        # Spatial attention layers\n        self.conv = Conv2D(1, kernel_size=7, padding='same', activation='sigmoid')\n\n    def build(self, input_shape):\n        # Ensure variables are built once\n        self.reshape_layer = Reshape((1, 1, self.channels))\n        self.concat_layer = Concatenate(axis=-1)\n        self.multiply_layer = Multiply()\n\n    def call(self, inputs):\n        # Channel Attention\n        avg_out = self.global_avg_pool(inputs)\n        max_out = self.global_max_pool(inputs)\n        avg_out = self.fc2(self.fc1(self.reshape_layer(avg_out)))\n        max_out = self.fc2(self.fc1(self.reshape_layer(max_out)))\n        channel_attention = self.multiply_layer([inputs, avg_out + max_out])\n\n        # Spatial Attention\n        avg_pool = tf.reduce_mean(channel_attention, axis=-1, keepdims=True)\n        max_pool = tf.reduce_max(channel_attention, axis=-1, keepdims=True)\n        spatial_attention = self.conv(self.concat_layer([avg_pool, max_pool]))\n        return self.multiply_layer([channel_attention, spatial_attention])\n\n    def get_config(self):\n        config = super(CBAM, self).get_config()\n        config.update({\n            \"channels\": self.channels,\n            \"reduction_ratio\": self.reduction_ratio\n        })\n        return config\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:38.898343Z","iopub.execute_input":"2024-12-09T21:09:38.898762Z","iopub.status.idle":"2024-12-09T21:09:38.916318Z","shell.execute_reply.started":"2024-12-09T21:09:38.898723Z","shell.execute_reply":"2024-12-09T21:09:38.915007Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"cnn = keras.models.load_model('/kaggle/input/120724-model/keras/default/1/120724.keras', custom_objects={'CBAM': CBAM})\ncnn.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:38.918022Z","iopub.execute_input":"2024-12-09T21:09:38.918496Z","iopub.status.idle":"2024-12-09T21:09:41.643385Z","shell.execute_reply.started":"2024-12-09T21:09:38.918444Z","shell.execute_reply":"2024-12-09T21:09:41.642215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict test images","metadata":{}},{"cell_type":"code","source":"test_preds = cnn.predict(test_loader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:09:41.645143Z","iopub.execute_input":"2024-12-09T21:09:41.645495Z","iopub.status.idle":"2024-12-09T21:17:09.437810Z","shell.execute_reply.started":"2024-12-09T21:09:41.645461Z","shell.execute_reply":"2024-12-09T21:17:09.436422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save predictions to csv","metadata":{}},{"cell_type":"code","source":"# Convert probabilities to binary labels (Malignant: 1 if prob > 0.5, else 0)\npredicted_labels = (test_preds > 0.5).astype(int).flatten()\n\n# Load sample submission file and assign predictions\nsubmission_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\n# Assemble submission file\nsubmission = pd.DataFrame({'id': submission_df['id'], 'label': predicted_labels})\n\n# Save the DataFrame to a CSV file for submission\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:47:30.147733Z","iopub.execute_input":"2024-12-09T21:47:30.148219Z","iopub.status.idle":"2024-12-09T21:47:30.338425Z","shell.execute_reply.started":"2024-12-09T21:47:30.148180Z","shell.execute_reply":"2024-12-09T21:47:30.337237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:47:34.958549Z","iopub.execute_input":"2024-12-09T21:47:34.959524Z","iopub.status.idle":"2024-12-09T21:47:34.973498Z","shell.execute_reply.started":"2024-12-09T21:47:34.959467Z","shell.execute_reply":"2024-12-09T21:47:34.971772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate frequency distribution\nfrequency_distribution = (submission.label.value_counts() / len(submission)).to_frame()\n\n# Plotting the frequency distribution as a bar chart\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']  # light green for benign, light red for malignant\n\n# Plotting bar chart with specified colors\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T21:47:37.507848Z","iopub.execute_input":"2024-12-09T21:47:37.508289Z","iopub.status.idle":"2024-12-09T21:47:37.789340Z","shell.execute_reply.started":"2024-12-09T21:47:37.508248Z","shell.execute_reply":"2024-12-09T21:47:37.787896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}