{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Histopathological Cancer Detection Project\n#Copy & Edit from MWV Final Project Training Model 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:46:50.390174Z","iopub.execute_input":"2025-10-30T16:46:50.390414Z","iopub.status.idle":"2025-10-30T16:46:50.393647Z","shell.execute_reply.started":"2025-10-30T16:46:50.390395Z","shell.execute_reply":"2025-10-30T16:46:50.392972Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN Model 5: Transfer learning (EfficientNet), Data augmentation, and <50% sample of Data","metadata":{}},{"cell_type":"markdown","source":"## Import packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers, datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:46:53.980977Z","iopub.execute_input":"2025-10-30T16:46:53.981778Z","iopub.status.idle":"2025-10-30T16:46:57.691630Z","shell.execute_reply.started":"2025-10-30T16:46:53.981748Z","shell.execute_reply":"2025-10-30T16:46:57.691045Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## View Data, Distributions, and Images with labels","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:04.416367Z","iopub.execute_input":"2025-10-30T16:47:04.417230Z","iopub.status.idle":"2025-10-30T16:47:04.633875Z","shell.execute_reply.started":"2025-10-30T16:47:04.417200Z","shell.execute_reply":"2025-10-30T16:47:04.633151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:09.349957Z","iopub.execute_input":"2025-10-30T16:47:09.350483Z","iopub.status.idle":"2025-10-30T16:47:09.360024Z","shell.execute_reply.started":"2025-10-30T16:47:09.350460Z","shell.execute_reply":"2025-10-30T16:47:09.359435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:12.774213Z","iopub.execute_input":"2025-10-30T16:47:12.774482Z","iopub.status.idle":"2025-10-30T16:47:12.792412Z","shell.execute_reply.started":"2025-10-30T16:47:12.774463Z","shell.execute_reply":"2025-10-30T16:47:12.791725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train.label.value_counts()/len(train.label)).to_frame().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:16.165337Z","iopub.execute_input":"2025-10-30T16:47:16.166090Z","iopub.status.idle":"2025-10-30T16:47:16.176289Z","shell.execute_reply.started":"2025-10-30T16:47:16.166059Z","shell.execute_reply":"2025-10-30T16:47:16.175581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Challenge: The image ids are used as filenames for the images, but the ids are missing the \".tif\" extension. \n#You will need to add a copy to the DataFrame to store the complete filename rather than just the id\ntrain['filenames'] = train['id']+'.tif'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:20.959075Z","iopub.execute_input":"2025-10-30T16:47:20.959372Z","iopub.status.idle":"2025-10-30T16:47:20.990466Z","shell.execute_reply.started":"2025-10-30T16:47:20.959349Z","shell.execute_reply":"2025-10-30T16:47:20.989847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:24.797464Z","iopub.execute_input":"2025-10-30T16:47:24.797768Z","iopub.status.idle":"2025-10-30T16:47:24.805561Z","shell.execute_reply.started":"2025-10-30T16:47:24.797744Z","shell.execute_reply":"2025-10-30T16:47:24.804940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_path = '/kaggle/input/histopathologic-cancer-detection/train'\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize = (6,6))\n\nfor i in range(len(sample)):\n    img = mpimg.imread(f'{train_images_path}/{sample.filenames[i]}')\n    #label = sample.label\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.title(sample.label[i])\n    plt.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:28.965814Z","iopub.execute_input":"2025-10-30T16:47:28.966444Z","iopub.status.idle":"2025-10-30T16:47:29.962358Z","shell.execute_reply.started":"2025-10-30T16:47:28.966420Z","shell.execute_reply":"2025-10-30T16:47:29.961525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sample the Data to make training more efficient","metadata":{}},{"cell_type":"code","source":"SS = 50000\nRS = 10\n\npositives = train[train['label']==1].sample(SS, random_state = RS)\nnegatives = train[train['label']==0].sample(SS, random_state = SS)\n\nnew_train = pd.concat([positives,negatives], axis = 0).reset_index(drop = True)\nnew_train = shuffle(new_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:37.294814Z","iopub.execute_input":"2025-10-30T16:47:37.295120Z","iopub.status.idle":"2025-10-30T16:47:37.378136Z","shell.execute_reply.started":"2025-10-30T16:47:37.295099Z","shell.execute_reply":"2025-10-30T16:47:37.377530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:42.678410Z","iopub.execute_input":"2025-10-30T16:47:42.679193Z","iopub.status.idle":"2025-10-30T16:47:42.687287Z","shell.execute_reply.started":"2025-10-30T16:47:42.679145Z","shell.execute_reply":"2025-10-30T16:47:42.686566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(new_train.label.value_counts()/len(new_train)).to_frame().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:47.757867Z","iopub.execute_input":"2025-10-30T16:47:47.758597Z","iopub.status.idle":"2025-10-30T16:47:47.767743Z","shell.execute_reply.started":"2025-10-30T16:47:47.758572Z","shell.execute_reply":"2025-10-30T16:47:47.767099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train_Test_Split","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(new_train, test_size = .2, random_state = 10, stratify = new_train.label)\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:52.776334Z","iopub.execute_input":"2025-10-30T16:47:52.777088Z","iopub.status.idle":"2025-10-30T16:47:52.822030Z","shell.execute_reply.started":"2025-10-30T16:47:52.777059Z","shell.execute_reply":"2025-10-30T16:47:52.821370Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Rescale Images with ImageDataGenerator","metadata":{}},{"cell_type":"code","source":"#Challenge: You will need to use an image data generator to load the files from disk. \ntrain_datagen = ImageDataGenerator(rescale = 1/255)\nval_datagen = ImageDataGenerator(rescale = 1/255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:47:56.726580Z","iopub.execute_input":"2025-10-30T16:47:56.727369Z","iopub.status.idle":"2025-10-30T16:47:56.731075Z","shell.execute_reply.started":"2025-10-30T16:47:56.727342Z","shell.execute_reply":"2025-10-30T16:47:56.730469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Make labels into strings","metadata":{}},{"cell_type":"code","source":"train_df['label'] = train_df['label'].astype(str)\nval_df['label'] = val_df['label'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:48:02.196590Z","iopub.execute_input":"2025-10-30T16:48:02.197298Z","iopub.status.idle":"2025-10-30T16:48:02.226465Z","shell.execute_reply.started":"2025-10-30T16:48:02.197270Z","shell.execute_reply":"2025-10-30T16:48:02.225729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Recommendation: I would recommend using Image Augmentation at some point.","metadata":{}},{"cell_type":"markdown","source":"## Create loaders using data augmentation\n* reduced batch size to see if it improves the models\n* horizontal and vertical flips\n* rotation\n* height and width shifts\n* Images sized 260x260 for the efficientnet model used","metadata":{}},{"cell_type":"code","source":"%%time\nbatch_size = 32\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    horizontal_flip = True,\n    vertical_flip = True,\n    height_shift_range = .12,\n    width_shift_range = .12,\n    rotation_range = 18,\n    target_size = (260,260)\n)\n\nval_loader = val_datagen.flow_from_dataframe(\n    dataframe = val_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (260,260)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:48:07.271840Z","iopub.execute_input":"2025-10-30T16:48:07.272126Z","iopub.status.idle":"2025-10-30T16:52:25.038687Z","shell.execute_reply.started":"2025-10-30T16:48:07.272107Z","shell.execute_reply":"2025-10-30T16:52:25.037923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVAL_STEPS = len(val_loader)\n\nprint(TR_STEPS)\nprint(VAL_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:52:49.534966Z","iopub.execute_input":"2025-10-30T16:52:49.535271Z","iopub.status.idle":"2025-10-30T16:52:49.539645Z","shell.execute_reply.started":"2025-10-30T16:52:49.535248Z","shell.execute_reply":"2025-10-30T16:52:49.538958Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create a model using the EfficientNetV2B2 model from keras as the foundation\n* Print the Architecture of the model\n* very little learning rate will be used to make sure the model doesn't over fit\n* More epochs will be used becasue of the efficiency of training compared to the other models\n* Input shape will be increased to 260x260 since that it what EfficientNetV2B2 uses\n* Training only the last 30 layers of the model\n* Switched to the Swich activation instead of the relu for the dense layers","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.activations import swish\nfrom tensorflow import keras\n\nbase_model_2=keras.applications.EfficientNetV2B2(\n    include_top=False,\n    weights=\"imagenet\",\n    input_tensor=None,\n    input_shape=(260,260,3),\n    pooling=None,\n    classes=2,\n    classifier_activation=\"Swish\",\n)\nbase_model_2.trainable = True\n\nfor layer in base_model_2.layers[:-30]:\n    layer.trainable = True\n\n\n\ncnn_model_5 = models.Sequential([\n    base_model_2,\n    layers.GlobalAveragePooling2D(),\n\n\n    Dense(512, activation = 'swish'),\n    Dropout(.4),\n    Dense(256, activation = 'swish'),\n    Dropout(.3),\n    Dense(128, activation = 'swish'),\n    Dropout(.2),\n    Dense(1, activation = 'sigmoid')\n])\ncnn_model_5.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:52:56.466443Z","iopub.execute_input":"2025-10-30T16:52:56.466734Z","iopub.status.idle":"2025-10-30T16:52:59.495503Z","shell.execute_reply.started":"2025-10-30T16:52:56.466711Z","shell.execute_reply":"2025-10-30T16:52:59.494789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate = 1e-5)\ncnn_model_5.compile(loss = 'binary_crossentropy', optimizer = opt, metrics = ['AUC'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:53:14.593179Z","iopub.execute_input":"2025-10-30T16:53:14.593685Z","iopub.status.idle":"2025-10-30T16:53:14.608867Z","shell.execute_reply.started":"2025-10-30T16:53:14.593660Z","shell.execute_reply":"2025-10-30T16:53:14.608324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Add callbacks for early stopping if the model doesn't improve\n## Add reduced learning rate if the learning stalls to see if it helps improve the model","metadata":{}},{"cell_type":"code","source":"early_stopping_callback = keras.callbacks.EarlyStopping(\n    monitor='val_AUC',      \n    patience=10,            \n    restore_best_weights=True, \n    mode='max',        \n    verbose=1               \n)\n\nlr_scheduler_callback = keras.callbacks.ReduceLROnPlateau(\n    monitor='val_AUC',      \n    factor=0.5,             \n    patience=5,             \n    min_lr=1e-8,            \n    mode='max',             \n    verbose=1               \n)\n\ncallbacks_list = [early_stopping_callback, lr_scheduler_callback]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:55:36.083213Z","iopub.execute_input":"2025-10-30T16:55:36.083773Z","iopub.status.idle":"2025-10-30T16:55:36.088183Z","shell.execute_reply.started":"2025-10-30T16:55:36.083744Z","shell.execute_reply":"2025-10-30T16:55:36.087463Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run the first training iteration\n* add callbacks\n* one running of 50 epochs with new learning rate changes","metadata":{}},{"cell_type":"code","source":"%%time\nh1 = cnn_model_5.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 10, # JWO was 50,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1,\n    callbacks = callbacks_list\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T16:56:18.587043Z","iopub.execute_input":"2025-10-30T16:56:18.587339Z","iopub.status.idle":"2025-10-30T18:21:10.557486Z","shell.execute_reply.started":"2025-10-30T16:56:18.587319Z","shell.execute_reply":"2025-10-30T18:21:10.556244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = h1.history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:21:28.202360Z","iopub.execute_input":"2025-10-30T18:21:28.202660Z","iopub.status.idle":"2025-10-30T18:21:28.206727Z","shell.execute_reply.started":"2025-10-30T18:21:28.202626Z","shell.execute_reply":"2025-10-30T18:21:28.206025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Display the model's performance\n* slowly but steadily increased in performance for both training and validation data\n* Validation plateaus while the training keeps improving showing some overfitting\n* Is pretty efficient learning, and does will after just a few epochs","metadata":{}},{"cell_type":"code","source":"epoch_range = range(1, len(history['loss'])+1)\nplt.figure(figsize = [12,5])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('Loss');plt.title(\"Loss\")\n\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['AUC'], label = 'Training')\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel(\"Epoch\");plt.ylabel(\"AUC\");plt.title(\"AUC\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:21:36.116794Z","iopub.execute_input":"2025-10-30T18:21:36.117560Z","iopub.status.idle":"2025-10-30T18:21:36.405034Z","shell.execute_reply.started":"2025-10-30T18:21:36.117535Z","shell.execute_reply":"2025-10-30T18:21:36.404380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save the model","metadata":{}},{"cell_type":"code","source":"import pickle\ncnn_model_5.save('Cancer_Detection_cnn_model_5.h5')\npickle.dump(history, open(f'Cancer_Detection_model_5.pk1', 'wb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:23:26.466137Z","iopub.execute_input":"2025-10-30T18:23:26.466774Z","iopub.status.idle":"2025-10-30T18:23:27.428031Z","shell.execute_reply.started":"2025-10-30T18:23:26.466748Z","shell.execute_reply":"2025-10-30T18:23:27.427230Z"}},"outputs":[],"execution_count":null}]}