{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Histopathological Cancer Detection Project","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:57:46.501924Z","iopub.execute_input":"2025-07-17T11:57:46.502134Z","iopub.status.idle":"2025-07-17T11:57:46.506428Z","shell.execute_reply.started":"2025-07-17T11:57:46.502116Z","shell.execute_reply":"2025-07-17T11:57:46.505820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CNN Model 3: Transfer learning, Data augmentation, and <50% sample of Data","metadata":{}},{"cell_type":"markdown","source":"## Import packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers, datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:57:46.507269Z","iopub.execute_input":"2025-07-17T11:57:46.507497Z","iopub.status.idle":"2025-07-17T11:58:00.631712Z","shell.execute_reply.started":"2025-07-17T11:57:46.507474Z","shell.execute_reply":"2025-07-17T11:58:00.631106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## View Data, Distributions, and Images with labels","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:00.633305Z","iopub.execute_input":"2025-07-17T11:58:00.633742Z","iopub.status.idle":"2025-07-17T11:58:00.989301Z","shell.execute_reply.started":"2025-07-17T11:58:00.633722Z","shell.execute_reply":"2025-07-17T11:58:00.988656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:00.989969Z","iopub.execute_input":"2025-07-17T11:58:00.990179Z","iopub.status.idle":"2025-07-17T11:58:01.010595Z","shell.execute_reply.started":"2025-07-17T11:58:00.990158Z","shell.execute_reply":"2025-07-17T11:58:01.009822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:01.011319Z","iopub.execute_input":"2025-07-17T11:58:01.011519Z","iopub.status.idle":"2025-07-17T11:58:01.042072Z","shell.execute_reply.started":"2025-07-17T11:58:01.011501Z","shell.execute_reply":"2025-07-17T11:58:01.041109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train.label.value_counts()/len(train.label)).to_frame().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:01.043034Z","iopub.execute_input":"2025-07-17T11:58:01.043370Z","iopub.status.idle":"2025-07-17T11:58:01.064605Z","shell.execute_reply.started":"2025-07-17T11:58:01.043340Z","shell.execute_reply":"2025-07-17T11:58:01.063841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['filenames'] = train['id']+'.tif'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:01.065427Z","iopub.execute_input":"2025-07-17T11:58:01.065731Z","iopub.status.idle":"2025-07-17T11:58:01.107801Z","shell.execute_reply.started":"2025-07-17T11:58:01.065711Z","shell.execute_reply":"2025-07-17T11:58:01.107050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:01.108643Z","iopub.execute_input":"2025-07-17T11:58:01.108933Z","iopub.status.idle":"2025-07-17T11:58:01.116804Z","shell.execute_reply.started":"2025-07-17T11:58:01.108905Z","shell.execute_reply":"2025-07-17T11:58:01.116080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_path = '/kaggle/input/histopathologic-cancer-detection/train'\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize = (6,6))\n\nfor i in range(len(sample)):\n    img = mpimg.imread(f'{train_images_path}/{sample.filenames[i]}')\n    #label = sample.label\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.title(sample.label[i])\n    plt.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:01.118902Z","iopub.execute_input":"2025-07-17T11:58:01.119136Z","iopub.status.idle":"2025-07-17T11:58:02.272478Z","shell.execute_reply.started":"2025-07-17T11:58:01.119118Z","shell.execute_reply":"2025-07-17T11:58:02.271641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sample the Data to make training more efficient","metadata":{}},{"cell_type":"code","source":"SS = 50000\nRS = 10\n\npositives = train[train['label']==1].sample(SS, random_state = RS)\nnegatives = train[train['label']==0].sample(SS, random_state = SS)\n\nnew_train = pd.concat([positives,negatives], axis = 0).reset_index(drop = True)\nnew_train = shuffle(new_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.273367Z","iopub.execute_input":"2025-07-17T11:58:02.273591Z","iopub.status.idle":"2025-07-17T11:58:02.370407Z","shell.execute_reply.started":"2025-07-17T11:58:02.273573Z","shell.execute_reply":"2025-07-17T11:58:02.369607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.371268Z","iopub.execute_input":"2025-07-17T11:58:02.371509Z","iopub.status.idle":"2025-07-17T11:58:02.379144Z","shell.execute_reply.started":"2025-07-17T11:58:02.371484Z","shell.execute_reply":"2025-07-17T11:58:02.378500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(new_train.label.value_counts()/len(new_train)).to_frame().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.379833Z","iopub.execute_input":"2025-07-17T11:58:02.380050Z","iopub.status.idle":"2025-07-17T11:58:02.399073Z","shell.execute_reply.started":"2025-07-17T11:58:02.380034Z","shell.execute_reply":"2025-07-17T11:58:02.398510Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train_Test_Split","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(new_train, test_size = .2, random_state = 10, stratify = new_train.label)\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.399753Z","iopub.execute_input":"2025-07-17T11:58:02.399969Z","iopub.status.idle":"2025-07-17T11:58:02.457711Z","shell.execute_reply.started":"2025-07-17T11:58:02.399950Z","shell.execute_reply":"2025-07-17T11:58:02.456904Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Rescale Images with ImageDataGenerator","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1/255)\nval_datagen = ImageDataGenerator(rescale = 1/255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.458710Z","iopub.execute_input":"2025-07-17T11:58:02.458968Z","iopub.status.idle":"2025-07-17T11:58:02.462905Z","shell.execute_reply.started":"2025-07-17T11:58:02.458950Z","shell.execute_reply":"2025-07-17T11:58:02.462351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Make labels into strings","metadata":{}},{"cell_type":"code","source":"train_df['label'] = train_df['label'].astype(str)\nval_df['label'] = val_df['label'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.463472Z","iopub.execute_input":"2025-07-17T11:58:02.463668Z","iopub.status.idle":"2025-07-17T11:58:02.501682Z","shell.execute_reply.started":"2025-07-17T11:58:02.463652Z","shell.execute_reply":"2025-07-17T11:58:02.501076Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create loaders using data augmentation\n* horizontal and vertical flips\n* rotation\n* height and width shifts\n* Images sized 64x64","metadata":{}},{"cell_type":"code","source":"%%time\nbatch_size = 96\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    horizontal_flip = True,\n    vertical_flip = True,\n    height_shift_range = .1,\n    width_shift_range = .1,\n    rotation_range = 15,\n    target_size = (96,96)\n)\n\nval_loader = val_datagen.flow_from_dataframe(\n    dataframe = val_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (96,96)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:58:02.502399Z","iopub.execute_input":"2025-07-17T11:58:02.502611Z","iopub.status.idle":"2025-07-17T12:03:45.817876Z","shell.execute_reply.started":"2025-07-17T11:58:02.502594Z","shell.execute_reply":"2025-07-17T12:03:45.817137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVAL_STEPS = len(val_loader)\n\nprint(TR_STEPS)\nprint(VAL_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:03:45.818725Z","iopub.execute_input":"2025-07-17T12:03:45.818971Z","iopub.status.idle":"2025-07-17T12:03:45.823083Z","shell.execute_reply.started":"2025-07-17T12:03:45.818953Z","shell.execute_reply":"2025-07-17T12:03:45.822466Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create a model using the ResNet101 model form keras as the foundation\n* Print the Architecture of the model\n* very little learning rate will be used to make sure the model doesn't over fit\n* More epochs will be used becasue of the efficiency of training compared to the other models","metadata":{}},{"cell_type":"code","source":"base_model_1 = tf.keras.applications.ResNet101(\n    input_shape=(96, 96, 3),\n    include_top=False,\n    weights='imagenet'\n)\nbase_model_1.trainable = True \n\n\n\ncnn_model_2 = models.Sequential([\n    base_model_1,\n    layers.GlobalAveragePooling2D(),\n\n\n    Dense(128, activation = 'relu'),\n    Dropout(.2),\n    Dense(64, activation = 'relu'),\n    Dropout(.2),\n    Dense(32, activation = 'relu'),\n    Dropout(.2),\n    Dense(1, activation = 'sigmoid')\n])\ncnn_model_2.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:03:45.823703Z","iopub.execute_input":"2025-07-17T12:03:45.823985Z","iopub.status.idle":"2025-07-17T12:03:51.376148Z","shell.execute_reply.started":"2025-07-17T12:03:45.823969Z","shell.execute_reply":"2025-07-17T12:03:51.375583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate = 1e-7)\ncnn_model_2.compile(loss = 'binary_crossentropy', optimizer = opt, metrics = ['AUC'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:03:51.376864Z","iopub.execute_input":"2025-07-17T12:03:51.377101Z","iopub.status.idle":"2025-07-17T12:03:51.391308Z","shell.execute_reply.started":"2025-07-17T12:03:51.377085Z","shell.execute_reply":"2025-07-17T12:03:51.390712Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run the first training iteration\n* very small learning rate","metadata":{}},{"cell_type":"code","source":"%%time\nh1 = cnn_model_2.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 50,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:03:51.391963Z","iopub.execute_input":"2025-07-17T12:03:51.392163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run the second training iteration\n* keep the same learning rate\n* stay with 50 epochs","metadata":{}},{"cell_type":"code","source":"history = h1.history","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nopt.learning_rate.assign(1e-7)\nh2 = cnn_model_2.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 50,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Display the model's performance\n* slowly but steadily increased in performance for both training and validation data\n* could keep training to improve performance, but balancing time and computational resources\n* Does extremely well using the whole 96x96 image size without taking too long to train","metadata":{}},{"cell_type":"code","source":"for k in history.keys():\n    history[k]+=h2.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\nplt.figure(figsize = [12,5])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('Loss');plt.title(\"Loss\")\n\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['AUC'], label = 'Training')\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel(\"Epoch\");plt.ylabel(\"AUC\");plt.title(\"AUC\")\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save the model","metadata":{}},{"cell_type":"code","source":"import pickle\ncnn_model_2.save('Cancer_Detection_cnn_model_2.h5')\npickle.dump(history, open(f'Cancer_Detection_model_2.pk1', 'wb'))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}