{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries and Packages","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, MaxPooling2D, Dropout, BatchNormalization, CenterCrop\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.callbacks import EarlyStopping, Callback\nfrom tensorflow.keras.metrics import AUC\nimport tifffile as tiff\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:55:26.392637Z","iopub.execute_input":"2024-11-24T23:55:26.393046Z","iopub.status.idle":"2024-11-24T23:55:41.649113Z","shell.execute_reply.started":"2024-11-24T23:55:26.392972Z","shell.execute_reply":"2024-11-24T23:55:41.647958Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Data paths\ntrain_images = '/kaggle/input/histopathologic-cancer-detection/train/'\ntest_images = '/kaggle/input/histopathologic-cancer-detection/test/'\nlabel_csv = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\nfull_df = pd.read_csv(label_csv)\nfull_df['id'] = full_df['id'] + '.tif'\nfull_df['label'] = full_df['label'].astype(str)\nprint(full_df.shape)\n\n# Frequency Distribution Plot\nfrequency_distribution = (full_df.label.value_counts() / len(full_df)).to_frame()\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\nplt.show()\n\n# Sample Images Plot\nsample_images = full_df.sample(16)\nfig, axes = plt.subplots(4, 4, figsize=(6, 6))\nfig.tight_layout(pad=1.0)\nfor i, ax in enumerate(axes.flat):\n    id = sample_images.iloc[i]['id']\n    label = sample_images.iloc[i]['label']\n    img = mpimg.imread(os.path.join(train_images, id))\n    ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Label: {label}\")\n    ax.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:56:54.401025Z","iopub.execute_input":"2024-11-24T23:56:54.402031Z","iopub.status.idle":"2024-11-24T23:56:56.611488Z","shell.execute_reply.started":"2024-11-24T23:56:54.401956Z","shell.execute_reply":"2024-11-24T23:56:56.610089Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split into Test and Training Datasets","metadata":{}},{"cell_type":"code","source":"# Split the data into train_df and valid_df with stratified sampling\ntrain_df, valid_df = train_test_split(\n    full_df,\n    test_size=0.2,  # 20% for validation\n    stratify=full_df['label'],  # Stratify by the label column to preserve proportions\n    random_state=42  # Set random seed for reproducibility\n)\ntrain_df['label'] = train_df['label'].astype(str)\nvalid_df['label'] = valid_df['label'].astype(str)\nprint(f\"Training set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(valid_df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:57:00.5603Z","iopub.execute_input":"2024-11-24T23:57:00.561071Z","iopub.status.idle":"2024-11-24T23:57:00.893145Z","shell.execute_reply.started":"2024-11-24T23:57:00.561033Z","shell.execute_reply":"2024-11-24T23:57:00.892018Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 1","metadata":{}},{"cell_type":"code","source":"# Function to apply random cropping for training images\ndef random_crop(img, crop_size=(32, 32)):\n    height, width = img.shape[:2]\n    dx, dy = crop_size\n    \n    if height < dx or width < dy:\n        raise ValueError(f\"Crop size {crop_size} is larger than image size {(height, width)}\")\n\n    x = np.random.randint(0, height - dx + 1)\n    y = np.random.randint(0, width - dy + 1)\n    \n    cropped_img = img[x:x+dx, y:y+dy]\n    if cropped_img.shape[:2] != crop_size:\n        raise ValueError(f\"Cropped image size {cropped_img.shape[:2]} does not match expected size {crop_size}\")\n    \n    return cropped_img\n\n# Define data generator with additional augmentations\ntrain_datagen = ImageDataGenerator(\n    rescale=1/255,\n    rotation_range=15,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    horizontal_flip=True,\n    zoom_range=0.1,\n    brightness_range=(0.8, 1.2)\n)\n\n# Data generator for validation, with minimal transformations\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T00:11:56.498637Z","iopub.execute_input":"2024-11-25T00:11:56.499065Z","iopub.status.idle":"2024-11-25T00:11:56.506736Z","shell.execute_reply.started":"2024-11-25T00:11:56.499026Z","shell.execute_reply":"2024-11-25T00:11:56.505708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Custom data loader to apply random cropping in addition to ImageDataGenerator\nclass CustomDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, generator, dataframe, directory, x_col, y_col, batch_size, target_size, crop_size, shuffle=True):\n        self.generator = generator.flow_from_dataframe(\n            dataframe=dataframe,\n            directory=directory,\n            x_col=x_col,\n            y_col=y_col,\n            class_mode='categorical',\n            batch_size=batch_size,\n            target_size=(96, 96),\n            shuffle=shuffle\n        )\n        self.crop_size = crop_size\n        self.target_size = target_size\n\n    def __len__(self):\n        return len(self.generator)\n\n    def __getitem__(self, idx):\n        batch_x, batch_y = self.generator[idx]\n        batch_x_cropped = np.array([random_crop(img, self.crop_size) for img in batch_x])\n        return batch_x_cropped, batch_y\n\n# Instantiate training and validation loaders\nBATCH_SIZE = 64\n\ntrain_loader = CustomDataGenerator(\n    generator=train_datagen,\n    dataframe=train_df,\n    directory=train_images,\n    x_col='id',\n    y_col='label',\n    batch_size=BATCH_SIZE,\n    target_size=(32, 32),\n    crop_size=(32, 32),\n    shuffle=True\n)\n\nvalid_loader = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=train_images,\n    x_col='id',\n    y_col='label',\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    target_size=(32, 32),\n    shuffle=False\n)\n\nTR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T00:11:59.769291Z","iopub.execute_input":"2024-11-25T00:11:59.769767Z","iopub.status.idle":"2024-11-25T00:13:58.22883Z","shell.execute_reply.started":"2024-11-25T00:11:59.769734Z","shell.execute_reply":"2024-11-25T00:13:58.22776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\nfrom tensorflow.keras.layers import LeakyReLU, GlobalAveragePooling2D\nfrom tensorflow.keras.regularizers import l2\n\ncnn1 = Sequential([\n    Conv2D(64, (3, 3), padding='same', input_shape=(32, 32, 3)),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    Conv2D(128, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    Conv2D(256, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.3),\n\n    Conv2D(512, (3, 3), padding='same'),\n    BatchNormalization(),\n    LeakyReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.3),\n\n    GlobalAveragePooling2D(),\n\n    Dense(256, activation='relu', kernel_regularizer=l2(0.001)),\n    BatchNormalization(),\n    Dropout(0.5),\n\n    Dense(128, activation='relu', kernel_regularizer=l2(0.001)),\n    BatchNormalization(),\n    Dropout(0.5),\n\n    Dense(2, activation='softmax')\n])\n\ncnn1.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T00:14:01.038035Z","iopub.execute_input":"2024-11-25T00:14:01.038398Z","iopub.status.idle":"2024-11-25T00:14:01.266194Z","shell.execute_reply.started":"2024-11-25T00:14:01.038364Z","shell.execute_reply":"2024-11-25T00:14:01.264809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the optimizer\noptimizer = Adam(learning_rate=0.0001)\ncnn1.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=[AUC(name='auc')])\n\n# Define a custom callback to print loss and AUC after each epoch\nclass PrintMetricsCallback(Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        train_loss = logs.get('loss')\n        train_auc = logs.get('auc')\n        val_loss = logs.get('val_loss')\n        val_auc = logs.get('val_auc')\n        print(f\"Epoch {epoch + 1}/{epochs} - Train Loss: {train_loss:.4f}, Train AUC: {train_auc:.4f} - Val Loss: {val_loss:.4f}, Val AUC: {val_auc:.4f}\")\n\n# Train Model 1\nepochs = 3\nearly_stopping = EarlyStopping(monitor='val_auc', patience=1, restore_best_weights=True)\n\nhistory1 = cnn1.fit(\n    train_loader,\n    validation_data=valid_loader,\n    epochs=epochs,\n    callbacks=[early_stopping, PrintMetricsCallback()],\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T00:14:08.424548Z","iopub.execute_input":"2024-11-25T00:14:08.424941Z","iopub.status.idle":"2024-11-25T01:28:09.863873Z","shell.execute_reply.started":"2024-11-25T00:14:08.424906Z","shell.execute_reply":"2024-11-25T01:28:09.862106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 2","metadata":{}},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\ncnn2 = Sequential([\n    keras.Input(shape=(96, 96, 3)),\n    CenterCrop(32, 32, 'channels_last'),\n\n    Conv2D(32, 3, padding='same', activation='selu'),\n    Dropout(.1),\n    MaxPooling2D(),\n\n    Conv2D(64, 3, padding='same', activation='selu'),\n    Dropout(.2),\n    MaxPooling2D(),\n\n    Conv2D(128, 3, padding='same', activation='selu'),\n    Dropout(.25),\n    MaxPooling2D(),\n\n    Conv2D(128, 3, padding='same', activation='selu'),\n    Dropout(.3),\n    MaxPooling2D(),\n\n    Conv2D(64, 3, padding='same', activation='selu'),\n    Dropout(.25),\n\n    Flatten(),\n    Dense(32, activation='selu'),\n    Dense(2, activation='sigmoid')  # Adjusted for binary classification\n])\n\ncnn2.summary()\n\n# Compile Model 2\noptimizer2 = keras.optimizers.Nadam(learning_rate=.001)\ncnn2.compile(optimizer=optimizer2, loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train Model 2\nearly_stopping = EarlyStopping(\n    monitor='val_accuracy',\n    patience=5,\n    min_delta=0.001,\n    mode='max'\n)\n\nhistory2 = cnn2.fit(\n    train_loader,\n    validation_data=valid_loader,\n    epochs=3,\n    verbose=1,\n    callbacks=[early_stopping]\n)\n\nh1 = pd.DataFrame(history2.history)\nh1ep = history2.epoch\ntf.keras.backend.set_value(cnn2.optimizer.learning_rate, .0001)\nhistory2 = cnn2.fit(train_loader, validation_data=valid_loader, epochs=5, verbose=1, callbacks=[early_stopping])\n\nhistory2.epoch = [x+history2.epoch[-1] for x in history2.epoch]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T01:28:56.267723Z","iopub.execute_input":"2024-11-25T01:28:56.268251Z","iopub.status.idle":"2024-11-25T03:03:51.380751Z","shell.execute_reply.started":"2024-11-25T01:28:56.268205Z","shell.execute_reply":"2024-11-25T03:03:51.378052Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 3","metadata":{}},{"cell_type":"code","source":"import cv2  # OpenCV library","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:04:11.207747Z","iopub.execute_input":"2024-11-25T03:04:11.20839Z","iopub.status.idle":"2024-11-25T03:04:11.471292Z","shell.execute_reply.started":"2024-11-25T03:04:11.208339Z","shell.execute_reply":"2024-11-25T03:04:11.470379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Open Cv Preprocess\ndef opencv_preprocess(image_path, target_size=(32, 32)):\n    image = cv2.imread(image_path)\n    if image is None:\n        raise ValueError(f\"Unable to load image at {image_path}\")\n    image = cv2.resize(image, target_size)\n    image = image / 255.0  # Normalize to [0, 1]\n    return image\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:04:12.906518Z","iopub.execute_input":"2024-11-25T03:04:12.90741Z","iopub.status.idle":"2024-11-25T03:04:12.912976Z","shell.execute_reply.started":"2024-11-25T03:04:12.90737Z","shell.execute_reply":"2024-11-25T03:04:12.911716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Modify the __getitem__ method in CustomDataGenerator to include OpenCV preprocessing\nclass CustomDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, generator, dataframe, directory, x_col, y_col, batch_size, target_size, crop_size, shuffle=True):\n        self.generator = generator.flow_from_dataframe(\n            dataframe=dataframe,\n            directory=directory,\n            x_col=x_col,\n            y_col=y_col,\n            class_mode='categorical',\n            batch_size=batch_size,\n            target_size=(32, 32),\n            shuffle=shuffle\n        )\n        self.crop_size = crop_size\n        self.target_size = target_size\n        self.directory = directory\n\n    def __len__(self):\n        return len(self.generator)\n\n    def __getitem__(self, idx):\n        batch_x, batch_y = self.generator[idx]\n        batch_x_opencv = np.array([opencv_preprocess(os.path.join(self.directory, img)) for img in batch_x])\n        batch_x_cropped = np.array([random_crop(img, self.crop_size) for img in batch_x_opencv])\n        return batch_x_cropped, batch_y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:04:15.855973Z","iopub.execute_input":"2024-11-25T03:04:15.856413Z","iopub.status.idle":"2024-11-25T03:04:15.864668Z","shell.execute_reply.started":"2024-11-25T03:04:15.856377Z","shell.execute_reply":"2024-11-25T03:04:15.863443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\ncnn3 = Sequential([\n    Conv2D(32, (3, 3), padding='same', input_shape=(32, 32, 3)),\n    BatchNormalization(),\n    tf.keras.layers.ReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    Conv2D(64, (3, 3), padding='same'),\n    BatchNormalization(),\n    tf.keras.layers.ReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.25),\n\n    Conv2D(128, (3, 3), padding='same'),\n    BatchNormalization(),\n    tf.keras.layers.ReLU(),\n    MaxPooling2D(pool_size=(2, 2)),\n    Dropout(0.3),\n\n    Flatten(),\n    Dense(128, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.5),\n\n    Dense(2, activation='softmax')\n])\n\ncnn3.summary()\n\n# Compile the model\noptimizer3 = Adam(learning_rate=0.0001)\ncnn3.compile(optimizer=optimizer3, loss='categorical_crossentropy', metrics=[AUC(name='auc')])\n\n# Define early stopping callback\nearly_stopping3 = EarlyStopping(monitor='val_auc', patience=1, restore_best_weights=True)\n\n# Train the model\nhistory3 = cnn3.fit(\n    train_loader,\n    validation_data=valid_loader,\n    epochs=3,\n    callbacks=[early_stopping3],\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:04:19.994763Z","iopub.execute_input":"2024-11-25T03:04:19.995194Z","iopub.status.idle":"2024-11-25T03:41:41.988776Z","shell.execute_reply.started":"2024-11-25T03:04:19.995158Z","shell.execute_reply":"2024-11-25T03:41:41.987838Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Plot Models","metadata":{}},{"cell_type":"code","source":"# Function to plot training history\ndef plot_history(history, title=''):\n    plt.figure(figsize=(12, 4))\n\n    # Plot training & validation loss values\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title(f'{title} Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend(['Train', 'Validation'], loc='upper right')\n\n    # Plot training & validation AUC values\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['auc'])\n    plt.plot(history.history['val_auc'])\n    plt.title(f'{title} AUC')\n    plt.xlabel('Epoch')\n    plt.ylabel('AUC')\n    plt.legend(['Train', 'Validation'], loc='lower right')\n\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:42:42.212853Z","iopub.execute_input":"2024-11-25T03:42:42.213286Z","iopub.status.idle":"2024-11-25T03:42:42.221799Z","shell.execute_reply.started":"2024-11-25T03:42:42.213253Z","shell.execute_reply":"2024-11-25T03:42:42.220563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot history for Model 1\nplot_history(history1, title='Model 1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:42:44.656976Z","iopub.execute_input":"2024-11-25T03:42:44.657409Z","iopub.status.idle":"2024-11-25T03:42:45.076277Z","shell.execute_reply.started":"2024-11-25T03:42:44.657373Z","shell.execute_reply":"2024-11-25T03:42:45.075172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot history for Model 2\nplot_history(history2, title='Model 2')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:42:47.850936Z","iopub.execute_input":"2024-11-25T03:42:47.85138Z","iopub.status.idle":"2024-11-25T03:42:48.362486Z","shell.execute_reply.started":"2024-11-25T03:42:47.851342Z","shell.execute_reply":"2024-11-25T03:42:48.361055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot history for Model 3 \nplot_history(history3, title='Model 3')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:42:51.719547Z","iopub.execute_input":"2024-11-25T03:42:51.719944Z","iopub.status.idle":"2024-11-25T03:42:52.158062Z","shell.execute_reply.started":"2024-11-25T03:42:51.719909Z","shell.execute_reply":"2024-11-25T03:42:52.157053Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Models","metadata":{}},{"cell_type":"code","source":"# Save Model 1\ncnn1.save('/kaggle/working/cnn_wk5_model.1.keras')\npickle.dump(history1.history, open('cnn_history_model1_v01.pkl', 'wb'))\n\n# Save Model 2\ncnn2.save('/kaggle/working/cnn_wk5_model.2.keras')\npickle.dump(history2.history, open('cnn_history_model2_v02.pkl', 'wb'))\n\n# Save Model 3\ncnn3.save('/kaggle/working/cnn_wk5_model.3.keras')\npickle.dump(history3.history, open('cnn_history_model3_v03.pkl', 'wb'))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T03:44:00.532172Z","iopub.execute_input":"2024-11-25T03:44:00.532622Z","iopub.status.idle":"2024-11-25T03:44:00.848828Z","shell.execute_reply.started":"2024-11-25T03:44:00.532584Z","shell.execute_reply":"2024-11-25T03:44:00.847871Z"}},"outputs":[],"execution_count":null}]}