{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30589,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nimport os\nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential, load_model\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import RMSprop, SGD\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom skimage import io\nimport tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:06:58.301493Z","iopub.execute_input":"2024-01-17T09:06:58.301874Z","iopub.status.idle":"2024-01-17T09:07:11.271518Z","shell.execute_reply.started":"2024-01-17T09:06:58.301832Z","shell.execute_reply":"2024-01-17T09:07:11.270422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"The train_labels.csv data is quite clean and contains an image id along with the label. No null counts were found and no cleaning is required.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")\n\nprint(df.head())\n\nprint(\"\\n----- Null counts -----\")\nnull_counts = df.isna().sum()\nprint(null_counts)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:32:43.319857Z","iopub.execute_input":"2024-01-17T09:32:43.320564Z","iopub.status.idle":"2024-01-17T09:32:43.588073Z","shell.execute_reply.started":"2024-01-17T09:32:43.320526Z","shell.execute_reply":"2024-01-17T09:32:43.587111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of tumor or no tumor is inbalanced. Yet, there is so much data available ~210000 images this will have little impact on this lab as for time's sake we will take a subset of these images for the model to save on training time.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom collections import Counter\nimport matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame\nlabels, counts = zip(*Counter(df[\"label\"]).items())\n\ntotal_images = sum(counts)\npercentages = [count / total_images * 100 for count in counts]\n\nplt.figure(figsize=(10, 6), dpi = 300)\nbars = plt.bar(labels, counts, color=[\"green\", \"red\"])\n\n# Display percentages on top of bars\nfor bar, percentage in zip(bars, percentages):\n    plt.text(bar.get_x() + bar.get_width() / 2 - 0.1, bar.get_height() + 0.1,\n             f'{percentage:.1f}%', fontsize=10, color='black')\n\nplt.xlabel(\"Label (0 = No Tumor, 1 = Tumor)\")\nplt.ylabel(\"Image Count\")\nplt.title(\"Distribution of Labels in the Data\")\nplt.xticks(labels)\nplt.grid(axis='y')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:32:48.990326Z","iopub.execute_input":"2024-01-17T09:32:48.991253Z","iopub.status.idle":"2024-01-17T09:32:49.551768Z","shell.execute_reply.started":"2024-01-17T09:32:48.99119Z","shell.execute_reply":"2024-01-17T09:32:49.550794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at some of the images with either of the two labels. There is no noteable difference between the two categories.","metadata":{}},{"cell_type":"code","source":"num_images_to_display = 3\n\ndf[\"path\"] = df[\"id\"].apply(lambda x: os.path.join(\"/kaggle/input/histopathologic-cancer-detection/train\", str(x) + \".tif\"))\nimages_with_label_0 = df[df[\"label\"] == 0]\nimages_with_label_1 = df[df[\"label\"] == 1]\nplt.figure(figsize=(10, 6), dpi = 300)\nfor i in range(num_images_to_display):\n    image = plt.imread(images_with_label_0[\"path\"].iloc[i])\n\n    plt.subplot(2, num_images_to_display, i+1)\n    plt.imshow(image)\n    plt.axis('off')\n    plt.title(\"Non-Tumor\")\n\nfor i in range(num_images_to_display):\n    image = plt.imread(images_with_label_1[\"path\"].iloc[i])\n\n    plt.subplot(2, num_images_to_display, num_images_to_display + i + 1)\n    plt.imshow(image)\n    plt.axis('off')\n    plt.title(\"Tumor\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:33:53.522709Z","iopub.execute_input":"2024-01-17T09:33:53.523471Z","iopub.status.idle":"2024-01-17T09:33:54.917706Z","shell.execute_reply.started":"2024-01-17T09:33:53.523434Z","shell.execute_reply":"2024-01-17T09:33:54.916769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\n# Function for image segmentation\ndef segment_image(image_path):\n    # Read the image\n    image = cv2.imread(image_path)\n\n    # Convert the image to grayscale\n    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    # Apply thresholding to obtain a binary mask\n    _, thresh = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY)\n\n    # Find contours in the binary mask\n    contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n    # Draw the contours on a blank image\n    segmented_image = np.zeros_like(image)\n    cv2.drawContours(segmented_image, contours, -1, (0, 255, 0), 2)\n\n    return segmented_image\n\n# Example usage: Segment and display the first image with label 1\nimage_path_label_1 = images_with_label_1[\"path\"].iloc[0]\nsegmented_image = segment_image(image_path_label_1)\n\n# Display the original and segmented images\nplt.figure(figsize=(10, 5), dpi = 300)\n\nplt.subplot(1, 2, 1)\nplt.imshow(plt.imread(image_path_label_1))\nplt.title(\"Original Image\")\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nplt.imshow(cv2.cvtColor(segmented_image, cv2.COLOR_BGR2RGB))\nplt.title(\"Segmented Image\")\nplt.axis('off')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:35:31.893918Z","iopub.execute_input":"2024-01-17T09:35:31.894641Z","iopub.status.idle":"2024-01-17T09:35:32.375755Z","shell.execute_reply.started":"2024-01-17T09:35:31.894606Z","shell.execute_reply":"2024-01-17T09:35:32.37476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seem as if all these images are of the same width, height and channel count. But check these are all the same is worth while.","metadata":{}},{"cell_type":"code","source":"unique_dimensions = set()\nrandom_indices = np.random.randint(0, len(df), 1000)\nfor index in random_indices:\n    image = io.imread(df[\"path\"].iloc[index])\n    image_height, image_width, image_channels = image.shape\n    unique_dimensions.add((image_height, image_width, image_channels))\n\nprint(f\"Unique dimension: {unique_dimensions}\")\nprint(f\"Image Height: {image_height} pixels\")\nprint(f\"Image Width: {image_width} pixels\")\nprint(f\"Number of Channels (Depth): {image_channels}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:37:23.133952Z","iopub.execute_input":"2024-01-17T09:37:23.135128Z","iopub.status.idle":"2024-01-17T09:37:31.930584Z","shell.execute_reply.started":"2024-01-17T09:37:23.135068Z","shell.execute_reply":"2024-01-17T09:37:31.929631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Initialize lists to store dimensions\nheights, widths, channels = [], [], []\n\n# Generate random indices\nrandom_indices = np.random.randint(0, len(df), 1000)\n\n# Loop through random indices and collect image dimensions\nfor index in random_indices:\n    image = io.imread(df[\"path\"].iloc[index])\n    image_height, image_width, image_channels = image.shape\n    heights.append(image_height)\n    widths.append(image_width)\n    channels.append(image_channels)\n\n# Create a scatter plot\nplt.figure(figsize=(12, 8))\nplt.scatter(widths, heights, c=channels, cmap='viridis', alpha=0.7)\nplt.colorbar(label='Number of Channels')\nplt.title('Scatter Plot of Image Dimensions')\nplt.xlabel('Image Width (pixels)')\nplt.ylabel('Image Height (pixels)')\nplt.grid(True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:37:31.932186Z","iopub.execute_input":"2024-01-17T09:37:31.932483Z","iopub.status.idle":"2024-01-17T09:37:41.557132Z","shell.execute_reply.started":"2024-01-17T09:37:31.932456Z","shell.execute_reply":"2024-01-17T09:37:41.556109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To ensure we gain the benefit of exploring different model changes we can take a subset of all our samples by taking 10000 images with label 0 and 10000 images with label 1. We can then use this as our training data and split it into training and test data.","metadata":{}},{"cell_type":"code","source":"df[\"label\"] = df[\"label\"].astype(str)\ndf_0 = df[df[\"label\"] == \"0\"].sample(10000, random_state=42)\ndf_1 = df[df[\"label\"] == \"1\"].sample(10000, random_state=42)\ndf_subset = pd.concat([df_0, df_1], ignore_index=True)\n\ntrain_file_paths, test_file_paths, train_labels, test_labels = train_test_split(df_subset[\"path\"], df_subset[\"label\"], test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:38:23.499068Z","iopub.execute_input":"2024-01-17T09:38:23.50004Z","iopub.status.idle":"2024-01-17T09:38:23.634187Z","shell.execute_reply.started":"2024-01-17T09:38:23.500004Z","shell.execute_reply":"2024-01-17T09:38:23.633199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plotting the distribution of labels\nlabels, counts = zip(*Counter(df_subset[\"label\"]).items())\n\nplt.figure(figsize=(8, 5), dpi = 300)\nplt.bar(labels, counts, color=[\"green\", \"red\"])\nplt.xlabel(\"Label (0 = No Tumor, 1 = Tumor)\")\nplt.ylabel(\"Image Count\")\nplt.title(\"Distribution of Labels in the Subset Data\")\nplt.xticks(labels)\nplt.grid(axis='y')\n\n# Annotating with percentages\ntotal_images = len(df_subset)\nfor label, count in zip(labels, counts):\n    percentage = (count / total_images) * 100\n    plt.text(label, count + 100, f'{percentage:.2f}%', ha='center', va='bottom', fontsize=10, color='black')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:39:03.945277Z","iopub.execute_input":"2024-01-17T09:39:03.945642Z","iopub.status.idle":"2024-01-17T09:39:04.399341Z","shell.execute_reply.started":"2024-01-17T09:39:03.945609Z","shell.execute_reply":"2024-01-17T09:39:04.398443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The class ImageDataGenerator is a Keras library used with data augmentation and preprocessing of image data with deep learning tasks. We gain the following benefits:\n- Augmenting the training data with rotations, width and height shifts, shearing, zooming and flipping reduces the changes we overfit our data. \n- Pixel scaling is done to normalize the pixel data giving a boost to the covergence of the training process. \n- Efficient loading allows us to work directly directories of images or dataframe.\n\nWe will use the directory load method but will need to split our chosen subset of data into a train folder and test folder. So we will copy these images to these new folders.","metadata":{}},{"cell_type":"code","source":"# Create directory called train_data and copy the training data into it\ntrain_dir = \"train_data\"\nif os.path.exists(train_dir):\n    shutil.rmtree(train_dir)\nos.makedirs(train_dir)\nos.makedirs(os.path.join(train_dir, \"0\"))\nos.makedirs(os.path.join(train_dir, \"1\"))\nfor file_path, label in zip(train_file_paths, train_labels):\n    name = file_path.split(\"/\")[-1]\n    if label == \"0\":\n        shutil.copy2(file_path, os.path.join(train_dir, \"0\", name))\n    else:\n        shutil.copy2(file_path, os.path.join(train_dir, \"1\", name))\n\n# Create directory called test_data and copy the test data into it\ntest_dir = \"test_data\"\nif os.path.exists(test_dir):\n    shutil.rmtree(test_dir)\nos.makedirs(test_dir)\nos.makedirs(os.path.join(test_dir, \"0\"))\nos.makedirs(os.path.join(test_dir, \"1\"))\nfor file_path, label in zip(test_file_paths, test_labels):\n    name = file_path.split(\"/\")[-1]\n    if label == \"0\":\n        shutil.copy2(file_path, os.path.join(test_dir, \"0\", name))\n    else:\n        shutil.copy2(file_path, os.path.join(test_dir, \"1\", name))\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode=\"nearest\"\n)\ntest_datagen = ImageDataGenerator(rescale=1.0 / 255)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=train_dir,\n    target_size=(image_width, image_height),\n    batch_size=32,\n    class_mode=\"binary\",\n    color_mode=\"rgb\",\n    shuffle=True,\n    seed=42\n)\ntest_generator = test_datagen.flow_from_directory(\n    directory=test_dir,\n    target_size=(image_width, image_height),\n    batch_size=32,\n    class_mode=\"binary\",\n    color_mode=\"rgb\",\n    shuffle=False,\n    seed=42\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:07:26.366352Z","iopub.execute_input":"2024-01-17T09:07:26.366736Z","iopub.status.idle":"2024-01-17T09:11:14.120293Z","shell.execute_reply.started":"2024-01-17T09:07:26.3667Z","shell.execute_reply":"2024-01-17T09:11:14.119045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the distribution of labels in the training data\ntrain_labels_dist = [str(int(label)) for label in train_generator.classes]\ntrain_labels, train_counts = zip(*Counter(train_labels_dist).items())\n\nplt.figure(figsize=(8, 5), dpi = 300)\nplt.bar(train_labels, train_counts, color=[\"green\", \"red\"])\nplt.xlabel(\"Label (0 = No Tumor, 1 = Tumor)\")\nplt.ylabel(\"Image Count\")\nplt.title(\"Distribution of Labels in the Training Data\")\nplt.xticks(train_labels)\nplt.grid(axis='y')\n\n# Annotating with percentages\ntotal_train_images = len(train_labels_dist)\nfor label, count in zip(train_labels, train_counts):\n    percentage = (count / total_train_images) * 100\n    plt.text(label, count + 50, f'{percentage:.2f}%', ha='center', va='bottom', fontsize=10, color='black')\n\nplt.show()\n\n# Plotting the distribution of labels in the test data\ntest_labels_dist = [str(int(label)) for label in test_generator.classes]\ntest_labels, test_counts = zip(*Counter(test_labels_dist).items())\n\nplt.figure(figsize=(8, 5), dpi = 300)\nplt.bar(test_labels, test_counts, color=[\"green\", \"red\"])\nplt.xlabel(\"Label (0 = No Tumor, 1 = Tumor)\")\nplt.ylabel(\"Image Count\")\nplt.title(\"Distribution of Labels in the Test Data\")\nplt.xticks(test_labels)\nplt.grid(axis='y')\n\n# Annotating with percentages\ntotal_test_images = len(test_labels_dist)\nfor label, count in zip(test_labels, test_counts):\n    percentage = (count / total_test_images) * 100\n    plt.text(label, count + 20, f'{percentage:.2f}%', ha='center', va='bottom', fontsize=10, color='black')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:40:32.829604Z","iopub.execute_input":"2024-01-17T09:40:32.829983Z","iopub.status.idle":"2024-01-17T09:40:33.738733Z","shell.execute_reply.started":"2024-01-17T09:40:32.829949Z","shell.execute_reply":"2024-01-17T09:40:33.737741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameters\nlearning_rate = 0.001\noptimizer = RMSprop(learning_rate=learning_rate)\ntraining_epochs = 10\nbatch_size = 32\ndropout_rate = 0.5\n\nmodel1 = Sequential()\npool_size = (3, 3)\nfilter_size = (3, 3)\n\n# Convolutional layers\nmodel1.add(Conv2D(16, filter_size, activation='relu', input_shape=(image_width, image_height, image_channels)))\nmodel1.add(Conv2D(16, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\nmodel1.add(Conv2D(32, filter_size, activation='relu'))\nmodel1.add(Conv2D(32, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\nmodel1.add(Conv2D(64, filter_size, activation='relu'))\nmodel1.add(Conv2D(64, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\n# Convert to 1D vector\nmodel1.add(Flatten())\n\n# Classification layers\nmodel1.add(Dense(64, activation='sigmoid'))\nmodel1.add(Dropout(dropout_rate))\nmodel1.add(Dense(1, activation='sigmoid'))\n\n# Compile the model\nmodel1.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\ntrain_steps_per_epoch=train_generator.n//train_generator.batch_size\nvalidation_steps_per_epoch=test_generator.n//test_generator.batch_size\nhistory1 = model1.fit(\n    train_generator,\n    steps_per_epoch=train_steps_per_epoch,\n    epochs=training_epochs,\n    validation_data=test_generator,\n    validation_steps=validation_steps_per_epoch\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:11:14.123678Z","iopub.execute_input":"2024-01-17T09:11:14.124028Z","iopub.status.idle":"2024-01-17T09:20:47.475966Z","shell.execute_reply.started":"2024-01-17T09:11:14.123998Z","shell.execute_reply":"2024-01-17T09:20:47.475133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport os\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, confusion_matrix, roc_auc_score\nfrom sklearn.metrics import roc_curve, auc\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, UpSampling2D, concatenate\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:40:55.986617Z","iopub.execute_input":"2024-01-17T09:40:55.987009Z","iopub.status.idle":"2024-01-17T09:40:55.993686Z","shell.execute_reply.started":"2024-01-17T09:40:55.986979Z","shell.execute_reply":"2024-01-17T09:40:55.992516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Constants\ninput_shape = (512, 512, 3)\nbatch_size = 16\nepochs = 20\nlearning_rate = 1e-4","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:41:04.983948Z","iopub.execute_input":"2024-01-17T09:41:04.984647Z","iopub.status.idle":"2024-01-17T09:41:04.989229Z","shell.execute_reply.started":"2024-01-17T09:41:04.984614Z","shell.execute_reply":"2024-01-17T09:41:04.988276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Build a simple U-Net model\ndef build_unet(input_shape):\n    inputs = Input(input_shape)\n    \n    # Encoder\n    conv1 = Conv2D(64, 3, activation='relu', padding='same')(inputs)\n    conv1 = Conv2D(64, 3, activation='relu', padding='same')(conv1)\n    pool1 = MaxPooling2D(pool_size=(2, 2))(conv1)\n    \n    # Decoder\n    conv2 = Conv2D(64, 3, activation='relu', padding='same')(pool1)\n    conv2 = Conv2D(64, 3, activation='relu', padding='same')(conv2)\n    up1 = UpSampling2D(size=(2, 2))(conv2)\n    \n    # Output\n    output = Conv2D(1, 1, activation='sigmoid')(up1)\n    \n    model = Model(inputs=inputs, outputs=output)\n    model.compile(optimizer=Adam(lr=learning_rate), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:41:11.584247Z","iopub.execute_input":"2024-01-17T09:41:11.584622Z","iopub.status.idle":"2024-01-17T09:41:11.593118Z","shell.execute_reply.started":"2024-01-17T09:41:11.584591Z","shell.execute_reply":"2024-01-17T09:41:11.591889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Function to plot training history\ndef plot_history(history):\n    plt.figure(figsize=(12, 4))\n    \n    # Plot training & validation accuracy values\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['accuracy'])\n    plt.plot(history.history['val_accuracy'])\n    plt.title('Model accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend(['Train', 'Validation'], loc='upper left')\n\n    # Plot training & validation loss values\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('Model loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend(['Train', 'Validation'], loc='upper left')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:41:38.51582Z","iopub.execute_input":"2024-01-17T09:41:38.516328Z","iopub.status.idle":"2024-01-17T09:41:38.524602Z","shell.execute_reply.started":"2024-01-17T09:41:38.516293Z","shell.execute_reply":"2024-01-17T09:41:38.52342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n# Define the 3D U-Net model\ndef unet3d_model(input_shape=(512, 512, 32, 1)):\n    inputs = tf.keras.Input(input_shape)\n    \n    # Encoder\n    conv1 = layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same')(inputs)\n    conv1 = layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same')(conv1)\n    pool1 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv1)\n\n    conv2 = layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same')(pool1)\n    conv2 = layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same')(conv2)\n    pool2 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv2)\n\n    # Add more encoder and decoder blocks as needed\n\n    # Decoder\n    up3 = layers.UpSampling3D(size=(2, 2, 2))(pool2)\n    up3 = layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same')(up3)\n    up3 = layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same')(up3)\n\n    up2 = layers.UpSampling3D(size=(2, 2, 2))(up3)\n    up2 = layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same')(up2)\n    up2 = layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same')(up2)\n\n    # Output layer\n    output = layers.Conv3D(1, (1, 1, 1), activation='sigmoid')(up2)\n\n    model = models.Model(inputs, output)\n    return model\n\n# Instantiate the model\nmodel3d = unet3d_model()\n\n# Display the model summary\nmodel3d.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:56:17.366865Z","iopub.execute_input":"2024-01-17T09:56:17.367268Z","iopub.status.idle":"2024-01-17T09:56:17.586634Z","shell.execute_reply.started":"2024-01-17T09:56:17.367237Z","shell.execute_reply":"2024-01-17T09:56:17.585026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:57:33.445614Z","iopub.execute_input":"2024-01-17T09:57:33.445977Z","iopub.status.idle":"2024-01-17T09:57:33.450818Z","shell.execute_reply.started":"2024-01-17T09:57:33.445947Z","shell.execute_reply":"2024-01-17T09:57:33.449805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Define constants\ninput_shape = (96, 96, 3)\nbatch_size = 32\nepochs = 10","metadata":{"execution":{"iopub.status.busy":"2024-01-17T09:57:37.645809Z","iopub.execute_input":"2024-01-17T09:57:37.646197Z","iopub.status.idle":"2024-01-17T09:57:37.651099Z","shell.execute_reply.started":"2024-01-17T09:57:37.646163Z","shell.execute_reply":"2024-01-17T09:57:37.649884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define U-Net model\ndef unet_model(input_shape=(96, 96, 3)):\n    inputs = tf.keras.Input(input_shape)\n    \n    # Encoder\n    conv1 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(inputs)\n    conv1 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(conv1)\n    pool1 = layers.MaxPooling2D(pool_size=(2, 2))(conv1)\n\n    conv2 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(pool1)\n    conv2 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(conv2)\n    pool2 = layers.MaxPooling2D(pool_size=(2, 2))(conv2)\n\n    # Add more encoder and decoder blocks as needed\n\n    # Decoder\n    up3 = layers.UpSampling2D(size=(2, 2))(pool2)\n    up3 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(up3)\n    up3 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(up3)\n\n    up2 = layers.UpSampling2D(size=(2, 2))(up3)\n    up2 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(up2)\n    up2 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(up2)\n    # Output laye\n    # Output layer\n    output = layers.Conv2D(1, (1, 1), activation='sigmoid')(up2)\n\n    model = models.Model(inputs, output)\n    return model\n\n# Instantiate the model\nmodel2d = unet_model(input_shape)\n\n# Compile the model\nmodel2d.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T10:00:27.937801Z","iopub.execute_input":"2024-01-17T10:00:27.938203Z","iopub.status.idle":"2024-01-17T10:00:28.058552Z","shell.execute_reply.started":"2024-01-17T10:00:27.938171Z","shell.execute_reply":"2024-01-17T10:00:28.057586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom skimage import io\nfrom collections import Counter\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Model\nfrom keras.layers import Input, Conv2D, MaxPooling2D, UpSampling2D, concatenate\nfrom keras.optimizers import Adam\n\n# Set random seed for reproducibility\nnp.random.seed(42)\n\n# Load dataset\n# Replace this path with the correct path to your dataset\ndataset_path = '/kaggle/input/histopathologic-cancer-detection/'\n\n# Load data labels\ndf = pd.read_csv(os.path.join(dataset_path, 'train_labels.csv'))\n\n# Add the file extension to the 'id' column\ndf['id'] = df['id'].astype(str) + '.tif'\n\n# Create a 'path' column for convenience\ndf['path'] = df['id'].apply(lambda x: os.path.join(dataset_path, 'train', x))\n\n# Sample and display images from each class\nnum_images_to_display = 3\n\ndf[\"label\"] = df[\"label\"].astype(str)\ndf_0 = df[df[\"label\"] == \"0\"].sample(10000, random_state=42)\ndf_1 = df[df[\"label\"] == \"1\"].sample(10000, random_state=42)\ndf_subset = pd.concat([df_0, df_1], ignore_index=True)\n\ntrain_file_paths, test_file_paths, train_labels, test_labels = train_test_split(\n    df_subset[\"path\"], df_subset[\"label\"], test_size=0.2, random_state=42\n)\n\n# Set input image dimensions\nimage_width, image_height = 96, 96\ninput_shape = (image_width, image_height, 3)\n\n# Create U-Net model\ndef unet_model(input_size=(96, 96, 3)):\n    inputs = Input(input_size)\n\n    # Contracting Path\n    conv1 = Conv2D(64, (3, 3), activation='relu', padding='same')(inputs)\n    conv1 = Conv2D(64, (3, 3), activation='relu', padding='same')(conv1)\n    pool1 = MaxPooling2D(pool_size=(2, 2))(conv1)\n\n    conv2 = Conv2D(128, (3, 3), activation='relu', padding='same')(pool1)\n    conv2 = Conv2D(128, (3, 3), activation='relu', padding='same')(conv2)\n    pool2 = MaxPooling2D(pool_size=(2, 2))(conv2)\n\n    conv3 = Conv2D(256, (3, 3), activation='relu', padding='same')(pool2)\n    conv3 = Conv2D(256, (3, 3), activation='relu', padding='same')(conv3)\n    pool3 = MaxPooling2D(pool_size=(2, 2))(conv3)\n\n    # Bottom\n    conv4 = Conv2D(512, (3, 3), activation='relu', padding='same')(pool3)\n    conv4 = Conv2D(512, (3, 3), activation='relu', padding='same')(conv4)\n\n    # Expanding Path\n    up5 = concatenate([UpSampling2D(size=(2, 2))(conv4), conv3], axis=-1)\n    conv5 = Conv2D(256, (3, 3), activation='relu', padding='same')(up5)\n    conv5 = Conv2D(256, (3, 3), activation='relu', padding='same')(conv5)\n\n    up6 = concatenate([UpSampling2D(size=(2, 2))(conv5), conv2], axis=-1)\n    conv6 = Conv2D(128, (3, 3), activation='relu', padding='same')(up6)\n    conv6 = Conv2D(128, (3, 3), activation='relu', padding='same')(conv6)\n\n    up7 = concatenate([UpSampling2D(size=(2, 2))(conv6), conv1], axis=-1)\n    conv7 = Conv2D(64, (3, 3), activation='relu', padding='same')(up7)\n    conv7 = Conv2D(64, (3, 3), activation='relu', padding='same')(conv7)\n\n    # Output layer\n    output = Conv2D(1, (1, 1), activation='sigmoid')(conv7)\n\n    model = Model(inputs=inputs, outputs=output)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-01-17T10:05:51.863317Z","iopub.execute_input":"2024-01-17T10:05:51.864174Z","iopub.status.idle":"2024-01-17T10:05:52.92475Z","shell.execute_reply.started":"2024-01-17T10:05:51.864138Z","shell.execute_reply":"2024-01-17T10:05:52.923702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U efficientnet -qq","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:28:24.664564Z","iopub.execute_input":"2024-01-17T11:28:24.6654Z","iopub.status.idle":"2024-01-17T11:28:36.37104Z","shell.execute_reply.started":"2024-01-17T11:28:24.665363Z","shell.execute_reply":"2024-01-17T11:28:36.369636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pickle\nimport os\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import * \nimport efficientnet.tfkeras as efn\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:28:36.373238Z","iopub.execute_input":"2024-01-17T11:28:36.373577Z","iopub.status.idle":"2024-01-17T11:28:36.555633Z","shell.execute_reply.started":"2024-01-17T11:28:36.373541Z","shell.execute_reply":"2024-01-17T11:28:36.554668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:28:42.297817Z","iopub.execute_input":"2024-01-17T11:28:42.298605Z","iopub.status.idle":"2024-01-17T11:28:42.307538Z","shell.execute_reply.started":"2024-01-17T11:28:42.298566Z","shell.execute_reply":"2024-01-17T11:28:42.306497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:28:49.888858Z","iopub.execute_input":"2024-01-17T11:28:49.889628Z","iopub.status.idle":"2024-01-17T11:28:50.115799Z","shell.execute_reply.started":"2024-01-17T11:28:49.889591Z","shell.execute_reply":"2024-01-17T11:28:50.114811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:28:56.137427Z","iopub.execute_input":"2024-01-17T11:28:56.137778Z","iopub.status.idle":"2024-01-17T11:28:56.157853Z","shell.execute_reply.started":"2024-01-17T11:28:56.137752Z","shell.execute_reply":"2024-01-17T11:28:56.156552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.id = train.id + '.tif'\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:02.355532Z","iopub.execute_input":"2024-01-17T11:29:02.35592Z","iopub.status.idle":"2024-01-17T11:29:02.393973Z","shell.execute_reply.started":"2024-01-17T11:29:02.35587Z","shell.execute_reply":"2024-01-17T11:29:02.393002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:08.546535Z","iopub.execute_input":"2024-01-17T11:29:08.546898Z","iopub.status.idle":"2024-01-17T11:29:08.5565Z","shell.execute_reply.started":"2024-01-17T11:29:08.546867Z","shell.execute_reply":"2024-01-17T11:29:08.555564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train.label.value_counts() / len(train)).to_frame().sort_index().T\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:13.910237Z","iopub.execute_input":"2024-01-17T11:29:13.910615Z","iopub.status.idle":"2024-01-17T11:29:13.954063Z","shell.execute_reply.started":"2024-01-17T11:29:13.910585Z","shell.execute_reply":"2024-01-17T11:29:13.952637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train\"\n\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize=(6,6))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{row.id}')    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:21.745906Z","iopub.execute_input":"2024-01-17T11:29:21.746663Z","iopub.status.idle":"2024-01-17T11:29:22.602788Z","shell.execute_reply.started":"2024-01-17T11:29:21.746626Z","shell.execute_reply":"2024-01-17T11:29:22.601739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, random_state=1, stratify=train.label)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:29.780755Z","iopub.execute_input":"2024-01-17T11:29:29.781105Z","iopub.status.idle":"2024-01-17T11:29:30.106401Z","shell.execute_reply.started":"2024-01-17T11:29:29.781063Z","shell.execute_reply":"2024-01-17T11:29:30.105592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:39.828269Z","iopub.execute_input":"2024-01-17T11:29:39.828648Z","iopub.status.idle":"2024-01-17T11:29:39.833382Z","shell.execute_reply.started":"2024-01-17T11:29:39.828615Z","shell.execute_reply":"2024-01-17T11:29:39.832473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:29:46.368506Z","iopub.execute_input":"2024-01-17T11:29:46.369234Z","iopub.status.idle":"2024-01-17T11:31:04.936812Z","shell.execute_reply.started":"2024-01-17T11:29:46.369196Z","shell.execute_reply":"2024-01-17T11:31:04.935907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:31:04.938374Z","iopub.execute_input":"2024-01-17T11:31:04.938662Z","iopub.status.idle":"2024-01-17T11:31:04.943408Z","shell.execute_reply.started":"2024-01-17T11:31:04.938636Z","shell.execute_reply":"2024-01-17T11:31:04.942476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = efn.EfficientNetB0(input_shape=(96,96,3), include_top=False, weights='imagenet')\n\ncnn = Sequential([\n    base_model,\n    \n    Flatten(),\n    \n    Dense(64, activation='relu'),\n    Dropout(0.5),\n    Dense(32, activation='relu'),\n    Dropout(0.5),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:31:04.944445Z","iopub.execute_input":"2024-01-17T11:31:04.944719Z","iopub.status.idle":"2024-01-17T11:31:08.391227Z","shell.execute_reply.started":"2024-01-17T11:31:04.944695Z","shell.execute_reply":"2024-01-17T11:31:08.390526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:31:08.39501Z","iopub.execute_input":"2024-01-17T11:31:08.395305Z","iopub.status.idle":"2024-01-17T11:31:08.420605Z","shell.execute_reply.started":"2024-01-17T11:31:08.395277Z","shell.execute_reply":"2024-01-17T11:31:08.41991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 40,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:31:24.130663Z","iopub.execute_input":"2024-01-17T11:31:24.131526Z","iopub.status.idle":"2024-01-17T13:02:54.027505Z","shell.execute_reply.started":"2024-01-17T11:31:24.131489Z","shell.execute_reply":"2024-01-17T13:02:54.026556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T13:02:54.029379Z","iopub.execute_input":"2024-01-17T13:02:54.030074Z","iopub.status.idle":"2024-01-17T13:02:54.972251Z","shell.execute_reply.started":"2024-01-17T13:02:54.030036Z","shell.execute_reply":"2024-01-17T13:02:54.971357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1, h2])\nvis_training(history, start=10)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T11:31:08.477624Z","iopub.status.idle":"2024-01-17T11:31:08.477966Z","shell.execute_reply.started":"2024-01-17T11:31:08.477797Z","shell.execute_reply":"2024-01-17T11:31:08.477813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nfrom sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\n\n# Assuming you have true labels and predicted probabilities\ntrue_labels = valid_loader.classes  # Replace with your true labels\npredictions = cnn.predict(valid_loader)  # Replace with your predicted probabilities\n\n# Get predicted labels\npredicted_labels = np.argmax(predictions, axis=1)\n\n# Confusion Matrix\ncm = confusion_matrix(true_labels, predicted_labels)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['0', '1'], yticklabels=['0', '1'])\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n\n# AUC-ROC Curve\nfpr, tpr, thresholds = roc_curve(true_labels, predictions[:, 1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = {:.2f})'.format(roc_auc))\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc='lower right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T13:05:32.191939Z","iopub.execute_input":"2024-01-17T13:05:32.192307Z","iopub.status.idle":"2024-01-17T13:06:45.824941Z","shell.execute_reply.started":"2024-01-17T13:05:32.192277Z","shell.execute_reply":"2024-01-17T13:06:45.824057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\n# Assuming you have validation data and true labels\nvalidation_data, true_labels = valid_loader.next()\n\n# Get predicted probabilities\npredictions = cnn.predict(validation_data)\n\n# Get predicted labels\npredicted_labels = np.argmax(predictions, axis=1)\n\n# Calculate the number of subplots needed\nnum_subplots = min(len(true_labels), 16)\n\n# Plot the images with circles around tumors\nplt.figure(figsize=(12, 8))\nfor i in range(num_subplots):\n    image = (validation_data[i] * 255).astype(np.uint8)  # Assuming images were normalized\n    true_label = true_labels[i]\n    predicted_label = predicted_labels[i]\n    \n    # Calculate subplot indices\n    rows = 4\n    cols = min(num_subplots, 4)\n    subplot_index = i % 16 + 1\n    \n    plt.subplot(rows, cols, subplot_index)\n    plt.imshow(image)\n    \n    # Draw a red circle if tumor is present\n    if predicted_label == 1:\n        # Find contours in the binary mask\n        contours, _ = cv2.findContours(np.round(predictions[i]).astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        \n        # Draw a red circle around the tumor\n        for contour in contours:\n            (x, y), radius = cv2.minEnclosingCircle(contour)\n            center = (int(x), int(y))\n            radius = int(radius)\n            image = cv2.circle(image, center, radius, (255, 0, 0), 2)\n\n    plt.title(f'True: {true_label}, Predicted: {predicted_label}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T13:11:27.304431Z","iopub.execute_input":"2024-01-17T13:11:27.305272Z","iopub.status.idle":"2024-01-17T13:11:28.877815Z","shell.execute_reply.started":"2024-01-17T13:11:27.305235Z","shell.execute_reply":"2024-01-17T13:11:28.876805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\n# Assuming you have validation data and true labels\nvalidation_data, true_labels = valid_loader.next()\n\n# Get predicted probabilities\npredictions = cnn.predict(validation_data)\n\n# Get predicted labels\npredicted_labels = np.argmax(predictions, axis=1)\n\n# Calculate the number of subplots needed\nnum_subplots = min(len(true_labels), 16)\n\n# Plot the images with highlighted tumors\nplt.figure(figsize=(12, 8))\nfor i in range(num_subplots):\n    image = (validation_data[i] * 255).astype(np.uint8)  # Assuming images were normalized\n    true_label = true_labels[i]\n    predicted_label = predicted_labels[i]\n    \n    # Calculate subplot indices\n    rows = 4\n    cols = min(num_subplots, 4)\n    subplot_index = i % 16 + 1\n    \n    plt.subplot(rows, cols, subplot_index)\n    plt.imshow(image)\n    \n    # Highlight tumor if predicted label is 1\n    if predicted_label == 1:\n        # Reshape the predictions to obtain the 2D array\n        predictions_2d = predictions[i].reshape((image.shape[0], image.shape[1], 2))\n        \n        # Apply a mask to highlight the tumor region\n        tumor_mask = np.round(predictions_2d[:, :, 1]).astype(np.uint8)\n        tumor_highlight = cv2.bitwise_and(image, image, mask=tumor_mask)\n        image = cv2.addWeighted(image, 1.2, tumor_highlight, 0.8, 0)\n    \n    plt.title(f'True: {true_label}, Predicted: {predicted_label}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T13:46:48.466843Z","iopub.execute_input":"2024-01-17T13:46:48.4679Z","iopub.status.idle":"2024-01-17T13:46:48.984512Z","shell.execute_reply.started":"2024-01-17T13:46:48.467851Z","shell.execute_reply":"2024-01-17T13:46:48.983139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}