{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Histopathological Cancer Detection Project","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Base Line Histopathological Cancer Detection Project**","metadata":{}},{"cell_type":"markdown","source":"## Import packages being used","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\n\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers, datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T18:20:47.292030Z","iopub.execute_input":"2025-07-14T18:20:47.292276Z","iopub.status.idle":"2025-07-14T18:20:47.296617Z","shell.execute_reply.started":"2025-07-14T18:20:47.292259Z","shell.execute_reply":"2025-07-14T18:20:47.296111Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Read the csv file and view its contents including any NaN values, as well as final count distribution of positive and negative cancer detection","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:54.078021Z","iopub.execute_input":"2025-07-14T13:37:54.078296Z","iopub.status.idle":"2025-07-14T13:37:54.299034Z","shell.execute_reply.started":"2025-07-14T13:37:54.078273Z","shell.execute_reply":"2025-07-14T13:37:54.298372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:54.467840Z","iopub.execute_input":"2025-07-14T13:37:54.468098Z","iopub.status.idle":"2025-07-14T13:37:54.477636Z","shell.execute_reply.started":"2025-07-14T13:37:54.468078Z","shell.execute_reply":"2025-07-14T13:37:54.476998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum().to_frame().T #see if any values are blank","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:54.716170Z","iopub.execute_input":"2025-07-14T13:37:54.716441Z","iopub.status.idle":"2025-07-14T13:37:54.736014Z","shell.execute_reply.started":"2025-07-14T13:37:54.716420Z","shell.execute_reply":"2025-07-14T13:37:54.735392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train.label.value_counts()/len(train)).to_frame() #view the percentage of Cancer vs Non-Cancer distribution","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:54.927852Z","iopub.execute_input":"2025-07-14T13:37:54.928092Z","iopub.status.idle":"2025-07-14T13:37:54.936912Z","shell.execute_reply.started":"2025-07-14T13:37:54.928076Z","shell.execute_reply":"2025-07-14T13:37:54.936300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Add a new column to the train df to include the '.tif' extension to properly view the images**","metadata":{}},{"cell_type":"code","source":"train['filenames']= train['id']+'.tif'#add new column with the .tif extension to fit the training samples\n#train['label'] = train['label'].astype(str) #needed to change to a string for the batch loaders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:55.107601Z","iopub.execute_input":"2025-07-14T13:37:55.107869Z","iopub.status.idle":"2025-07-14T13:37:55.139415Z","shell.execute_reply.started":"2025-07-14T13:37:55.107850Z","shell.execute_reply":"2025-07-14T13:37:55.138646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head() #new dataframe that has the proper names of the files in filenames column","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:55.478806Z","iopub.execute_input":"2025-07-14T13:37:55.479059Z","iopub.status.idle":"2025-07-14T13:37:55.486268Z","shell.execute_reply.started":"2025-07-14T13:37:55.479038Z","shell.execute_reply":"2025-07-14T13:37:55.485650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#show images\ntrain_images_path = '/kaggle/input/histopathologic-cancer-detection/train'\nsample = train.sample(n=5).reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:55.656706Z","iopub.execute_input":"2025-07-14T13:37:55.656953Z","iopub.status.idle":"2025-07-14T13:37:55.666432Z","shell.execute_reply.started":"2025-07-14T13:37:55.656935Z","shell.execute_reply":"2025-07-14T13:37:55.665897Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Get the proper information to show the training images and their labels","metadata":{}},{"cell_type":"code","source":"sample.filenames[1], sample.label[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:57.646294Z","iopub.execute_input":"2025-07-14T13:37:57.646541Z","iopub.status.idle":"2025-07-14T13:37:57.651592Z","shell.execute_reply.started":"2025-07-14T13:37:57.646523Z","shell.execute_reply":"2025-07-14T13:37:57.650980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_path = '/kaggle/input/histopathologic-cancer-detection/train'\nsample = train.sample(n = 16).reset_index()\n\nplt.figure(figsize = (6,6))\n\nfor i in range(len(sample)):\n    img = mpimg.imread(f'{train_images_path}/{sample.filenames[i]}')\n    label = sample.label\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.title(sample.label[i])\n\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:37:57.833098Z","iopub.execute_input":"2025-07-14T13:37:57.833320Z","iopub.status.idle":"2025-07-14T13:37:58.691134Z","shell.execute_reply.started":"2025-07-14T13:37:57.833301Z","shell.execute_reply":"2025-07-14T13:37:58.690338Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **perform a train-test split for the training data. I chose to do a 75% to 25% train to validation data**","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(train, test_size = 0.25,random_state = 10,  stratify = train.label)\n\nprint(train_df.shape)\nprint(val_df.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:38:00.469227Z","iopub.execute_input":"2025-07-14T13:38:00.469504Z","iopub.status.idle":"2025-07-14T13:38:00.569255Z","shell.execute_reply.started":"2025-07-14T13:38:00.469484Z","shell.execute_reply":"2025-07-14T13:38:00.568629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Rescale the images for the ImageDataGenerator**","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1/255)\nval_datagen = ImageDataGenerator(rescale = 1/255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:38:00.951019Z","iopub.execute_input":"2025-07-14T13:38:00.951494Z","iopub.status.idle":"2025-07-14T13:38:00.954876Z","shell.execute_reply.started":"2025-07-14T13:38:00.951471Z","shell.execute_reply":"2025-07-14T13:38:00.954298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The loaders weren't properly getting the labels since they were int() So they had to be changed to str() values","metadata":{}},{"cell_type":"code","source":"val_df['label'] = val_df['label'].astype(str) #change label to string for the loader\ntrain_df['label'] = train_df['label'].astype(str)  #change label to string for the loaders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:38:01.314971Z","iopub.execute_input":"2025-07-14T13:38:01.315204Z","iopub.status.idle":"2025-07-14T13:38:01.368306Z","shell.execute_reply.started":"2025-07-14T13:38:01.315186Z","shell.execute_reply":"2025-07-14T13:38:01.367522Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Properly created the loaders to connect with the various paths in kaggle and obtained the steps to be taken during the fit method**","metadata":{}},{"cell_type":"code","source":"%%time\nbatch_size = 512 #pretty high batch size since the dataset is so large\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (32,32)\n)\n\nval_loader = val_datagen.flow_from_dataframe(\n    dataframe = val_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (32,32)\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:38:06.181718Z","iopub.execute_input":"2025-07-14T13:38:06.182024Z","iopub.status.idle":"2025-07-14T13:46:41.308029Z","shell.execute_reply.started":"2025-07-14T13:38:06.182003Z","shell.execute_reply":"2025-07-14T13:46:41.307441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVAL_STEPS = len(val_loader)\n\nprint(TR_STEPS)\nprint(VAL_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:46:41.308994Z","iopub.execute_input":"2025-07-14T13:46:41.309244Z","iopub.status.idle":"2025-07-14T13:46:41.313610Z","shell.execute_reply.started":"2025-07-14T13:46:41.309227Z","shell.execute_reply":"2025-07-14T13:46:41.312806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Created the CNN using 2 Convolution blocks with filters increasing, but keeping the architecture relatively simple**\nmain points: \n* filters increase from 16 ot 32 in the second block\n* Training took far too long with the 3rd block of 64 filters added\n* MaxPooling added along with .2 Dropout rate\n* ended with 4 dense layers reducing from 128 to 1\n* sigmoid was used for the last layer because of the 0,1 decision","metadata":{}},{"cell_type":"code","source":"cnn_baseline = Sequential([\n    Conv2D(filters = 16, kernel_size = [3,3], padding = 'same', activation = 'relu', input_shape = (32,32,3)),\n    Conv2D(filters = 16, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    Conv2D(filters = 16, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    MaxPooling2D(2,2),\n    Dropout(.2),\n    BatchNormalization(),\n\n    Conv2D(filters = 32, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    Conv2D(filters = 32, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    Conv2D(filters = 32, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    MaxPooling2D(2,2),\n    Dropout(.2),\n    BatchNormalization(),\n\n    #Conv2D(filters = 64, kernel_size = [5,5], padding = 'same', activation = 'relu'),\n    #Conv2D(filters = 64, kernel_size = [5,5], padding = 'same', activation = 'relu'),\n    #Conv2D(filters = 64, kernel_size = [3,3], padding = 'same', activation = 'relu'),\n    #MaxPooling2D(2,2),\n    #Dropout(.2),\n    #BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(128, activation = 'relu'),\n    Dense(64, activation = 'relu'),\n    Dense(32, activation = 'relu'),\n    Dense(1, activation = 'sigmoid')\n])\n\ncnn_baseline.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:46:41.314506Z","iopub.execute_input":"2025-07-14T13:46:41.314746Z","iopub.status.idle":"2025-07-14T13:46:42.496798Z","shell.execute_reply.started":"2025-07-14T13:46:41.314724Z","shell.execute_reply":"2025-07-14T13:46:42.496282Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Optimizer created, model compiled, then fit to the data**","metadata":{}},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate = .001)\ncnn_baseline.compile(loss = 'binary_crossentropy', optimizer = opt, metrics = ['AUC'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:46:42.497945Z","iopub.execute_input":"2025-07-14T13:46:42.498140Z","iopub.status.idle":"2025-07-14T13:46:42.509340Z","shell.execute_reply.started":"2025-07-14T13:46:42.498125Z","shell.execute_reply":"2025-07-14T13:46:42.508803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nh1 = cnn_baseline.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 10,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T13:46:42.510036Z","iopub.execute_input":"2025-07-14T13:46:42.510274Z","iopub.status.idle":"2025-07-14T16:05:20.483059Z","shell.execute_reply.started":"2025-07-14T13:46:42.510257Z","shell.execute_reply":"2025-07-14T16:05:20.482393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **The model improved pretty consistently with some jumps in the validation improvements**\n* *The model can still improve, so another training will be done with lowering the learning rate*","metadata":{}},{"cell_type":"code","source":"history = h1.history\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize = [12,5])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('Loss');plt.title('Loss')\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['AUC'], label = \"Training\")\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.title('AUC')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T16:07:17.561298Z","iopub.execute_input":"2025-07-14T16:07:17.561799Z","iopub.status.idle":"2025-07-14T16:07:17.860468Z","shell.execute_reply.started":"2025-07-14T16:07:17.561751Z","shell.execute_reply":"2025-07-14T16:07:17.859654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The training took roughly 2 hours and 20 minutes but it could go improve still!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T16:08:15.210741Z","iopub.execute_input":"2025-07-14T16:08:15.211421Z","iopub.status.idle":"2025-07-14T16:08:15.214907Z","shell.execute_reply.started":"2025-07-14T16:08:15.211397Z","shell.execute_reply":"2025-07-14T16:08:15.214205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt.learning_rate.assign(.0005)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T16:09:02.636165Z","iopub.execute_input":"2025-07-14T16:09:02.637038Z","iopub.status.idle":"2025-07-14T16:09:02.642978Z","shell.execute_reply.started":"2025-07-14T16:09:02.637006Z","shell.execute_reply":"2025-07-14T16:09:02.642316Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The model is still improving slightly, but it takes a while to train. Holding off from further training but will keep the same training structure of 20 epochs with reduced learning rate to compare with other models in time efficiency and performance. ","metadata":{}},{"cell_type":"code","source":"%%time\nh2 = cnn_baseline.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 10,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T16:10:50.543581Z","iopub.execute_input":"2025-07-14T16:10:50.544362Z","iopub.status.idle":"2025-07-14T18:20:46.841559Z","shell.execute_reply.started":"2025-07-14T16:10:50.544329Z","shell.execute_reply":"2025-07-14T18:20:46.840977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for k in history.keys():\n    history[k]+=h2.history[k]\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize = [12,5])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['AUC'], label = 'Training')\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T18:20:46.844355Z","iopub.execute_input":"2025-07-14T18:20:46.844601Z","iopub.status.idle":"2025-07-14T18:20:47.176592Z","shell.execute_reply.started":"2025-07-14T18:20:46.844582Z","shell.execute_reply":"2025-07-14T18:20:47.175894Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Saving the model. Perfoming on the test dataset to compare to other models","metadata":{}},{"cell_type":"code","source":"cnn_baseline.save('Cancer_Detection_cnn_baseline.h5')\npickle.dump(history, open(f'Cancer_Detection_Baseline.pk1', 'wb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T18:20:47.177483Z","iopub.execute_input":"2025-07-14T18:20:47.177785Z","iopub.status.idle":"2025-07-14T18:20:47.290739Z","shell.execute_reply.started":"2025-07-14T18:20:47.177742Z","shell.execute_reply":"2025-07-14T18:20:47.290236Z"}},"outputs":[],"execution_count":null}]}