{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Histopathological Cancer Detection Project with Data Augmentation and Sample Data","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:09:58.344323Z","iopub.execute_input":"2025-07-16T14:09:58.345096Z","iopub.status.idle":"2025-07-16T14:09:58.348301Z","shell.execute_reply.started":"2025-07-16T14:09:58.345069Z","shell.execute_reply":"2025-07-16T14:09:58.347675Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\n\nimport pickle\n\nimport tensorflow as tf\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers, datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:09:59.123937Z","iopub.execute_input":"2025-07-16T14:09:59.124588Z","iopub.status.idle":"2025-07-16T14:10:12.793272Z","shell.execute_reply.started":"2025-07-16T14:09:59.124561Z","shell.execute_reply":"2025-07-16T14:10:12.792422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## View Data, Distributions, and images","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:12.794517Z","iopub.execute_input":"2025-07-16T14:10:12.795003Z","iopub.status.idle":"2025-07-16T14:10:13.101579Z","shell.execute_reply.started":"2025-07-16T14:10:12.794983Z","shell.execute_reply":"2025-07-16T14:10:13.100789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.102408Z","iopub.execute_input":"2025-07-16T14:10:13.102598Z","iopub.status.idle":"2025-07-16T14:10:13.121485Z","shell.execute_reply.started":"2025-07-16T14:10:13.102584Z","shell.execute_reply":"2025-07-16T14:10:13.120733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum().to_frame().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.123533Z","iopub.execute_input":"2025-07-16T14:10:13.123845Z","iopub.status.idle":"2025-07-16T14:10:13.141031Z","shell.execute_reply.started":"2025-07-16T14:10:13.123821Z","shell.execute_reply":"2025-07-16T14:10:13.140386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train.label.value_counts()/len(train)).to_frame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.141808Z","iopub.execute_input":"2025-07-16T14:10:13.142014Z","iopub.status.idle":"2025-07-16T14:10:13.162208Z","shell.execute_reply.started":"2025-07-16T14:10:13.141997Z","shell.execute_reply":"2025-07-16T14:10:13.161488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['filenames'] = train['id']+'.tif'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.163323Z","iopub.execute_input":"2025-07-16T14:10:13.163582Z","iopub.status.idle":"2025-07-16T14:10:13.205080Z","shell.execute_reply.started":"2025-07-16T14:10:13.163556Z","shell.execute_reply":"2025-07-16T14:10:13.204365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.205869Z","iopub.execute_input":"2025-07-16T14:10:13.206099Z","iopub.status.idle":"2025-07-16T14:10:13.216007Z","shell.execute_reply.started":"2025-07-16T14:10:13.206082Z","shell.execute_reply":"2025-07-16T14:10:13.215360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_path = '/kaggle/input/histopathologic-cancer-detection/train'\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize = (6,6))\n\nfor i in range(len(sample)):\n    img = mpimg.imread(f'{train_images_path}/{sample.filenames[i]}')\n    label = sample.label\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.title(sample.label[i])\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:13.216697Z","iopub.execute_input":"2025-07-16T14:10:13.216943Z","iopub.status.idle":"2025-07-16T14:10:14.395037Z","shell.execute_reply.started":"2025-07-16T14:10:13.216917Z","shell.execute_reply":"2025-07-16T14:10:14.394302Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sample the Data to make training much more efficient","metadata":{}},{"cell_type":"code","source":"SS = 50000\nRS = 10\n\npositives =train[train['label']==1].sample(SS, random_state = RS)\nnegatives = train[train['label'] == 0].sample(SS, random_state = RS)\n\nnew_train = pd.concat([positives, negatives], axis = 0).reset_index(drop=True)\nnew_train = shuffle(new_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.395823Z","iopub.execute_input":"2025-07-16T14:10:14.396025Z","iopub.status.idle":"2025-07-16T14:10:14.467973Z","shell.execute_reply.started":"2025-07-16T14:10:14.396009Z","shell.execute_reply":"2025-07-16T14:10:14.467385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.469884Z","iopub.execute_input":"2025-07-16T14:10:14.470140Z","iopub.status.idle":"2025-07-16T14:10:14.477559Z","shell.execute_reply.started":"2025-07-16T14:10:14.470122Z","shell.execute_reply":"2025-07-16T14:10:14.476894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(new_train.label.value_counts()/len(new_train)).to_frame().T #this also gives an even distribution of data!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.478294Z","iopub.execute_input":"2025-07-16T14:10:14.478507Z","iopub.status.idle":"2025-07-16T14:10:14.499075Z","shell.execute_reply.started":"2025-07-16T14:10:14.478492Z","shell.execute_reply":"2025-07-16T14:10:14.498376Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Perform Train_test_split","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(new_train, test_size = .2, random_state = 10, stratify = new_train.label)\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.499691Z","iopub.execute_input":"2025-07-16T14:10:14.499854Z","iopub.status.idle":"2025-07-16T14:10:14.554694Z","shell.execute_reply.started":"2025-07-16T14:10:14.499840Z","shell.execute_reply":"2025-07-16T14:10:14.553913Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Rescale Images with ImageDataGenerator","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1/255)\nval_datagen = ImageDataGenerator(rescale = 1/255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.555362Z","iopub.execute_input":"2025-07-16T14:10:14.555546Z","iopub.status.idle":"2025-07-16T14:10:14.559382Z","shell.execute_reply.started":"2025-07-16T14:10:14.555532Z","shell.execute_reply":"2025-07-16T14:10:14.558459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Change labels to strings!","metadata":{}},{"cell_type":"code","source":"train_df['label'] = train_df['label'].astype(str)\nval_df['label'] = val_df['label'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.560272Z","iopub.execute_input":"2025-07-16T14:10:14.560522Z","iopub.status.idle":"2025-07-16T14:10:14.596554Z","shell.execute_reply.started":"2025-07-16T14:10:14.560499Z","shell.execute_reply":"2025-07-16T14:10:14.595945Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create loaders using data augmentation \n* (horizontal and vertical flips, rotation, and height and width shifts)\n* Images sized to 64x64 to start\n","metadata":{}},{"cell_type":"code","source":"%%time\nbatch_size = 64 #smaller due to smaller dataset\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (96,96),\n    horizontal_flip = True,\n    vertical_flip = True,\n    height_shift_range = .1,\n    width_shift_range  = .1,\n    rotation_range = 15\n)\n\nval_loader = val_datagen.flow_from_dataframe(\n    dataframe = val_df,\n    directory = train_images_path,\n    x_col = 'filenames',\n    y_col = 'label',\n    batch_size = batch_size,\n    seed = 10,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (96,96)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:10:14.597280Z","iopub.execute_input":"2025-07-16T14:10:14.597449Z","iopub.status.idle":"2025-07-16T14:16:24.702015Z","shell.execute_reply.started":"2025-07-16T14:10:14.597436Z","shell.execute_reply":"2025-07-16T14:16:24.701216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVAL_STEPS = len(val_loader)\n\nprint(TR_STEPS)\nprint(VAL_STEPS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:16:24.702944Z","iopub.execute_input":"2025-07-16T14:16:24.703485Z","iopub.status.idle":"2025-07-16T14:16:24.707411Z","shell.execute_reply.started":"2025-07-16T14:16:24.703455Z","shell.execute_reply":"2025-07-16T14:16:24.706893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create the Sequential Model with increasing filters and new input shape size","metadata":{}},{"cell_type":"code","source":"cnn_model_3 = Sequential([\n    Conv2D(32, (3,3), activation = 'relu', input_shape = (96,96,3)),\n    Conv2D(32, (3,3), activation = 'relu'),\n    Conv2D(32, (3,3), activation = 'relu'),\n\n    MaxPooling2D(2,2),\n    Dropout(.2),\n    BatchNormalization(),\n\n    Conv2D(64, (3,3), activation = 'relu'),\n    Conv2D(64, (3,3), activation = 'relu'),\n    Conv2D(64, (3,3), activation = 'relu'),\n\n    MaxPooling2D(2,2),\n    Dropout(.2),\n    BatchNormalization(),\n\n    Conv2D(128, (3,3), activation = 'relu'),\n    Conv2D(128, (3,3), activation = 'relu'),\n    Conv2D(128, (3,3), activation = 'relu'),\n\n    MaxPooling2D(2,2),\n    Dropout(.2),\n    BatchNormalization(),\n\n    Flatten(),\n\n    Dense(256, activation = 'relu'),\n    Dropout(.2),\n    Dense(128, activation = 'relu'),\n    Dropout(.2),\n    Dense(32, activation = 'relu'),\n    Dropout(.2),\n    Dense(1, activation = 'sigmoid')\n\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:27:34.854733Z","iopub.execute_input":"2025-07-16T14:27:34.855450Z","iopub.status.idle":"2025-07-16T14:27:34.980688Z","shell.execute_reply.started":"2025-07-16T14:27:34.855424Z","shell.execute_reply":"2025-07-16T14:27:34.980140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn_model_3.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:27:35.449074Z","iopub.execute_input":"2025-07-16T14:27:35.449330Z","iopub.status.idle":"2025-07-16T14:27:35.475995Z","shell.execute_reply.started":"2025-07-16T14:27:35.449313Z","shell.execute_reply":"2025-07-16T14:27:35.475421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate = .001)\ncnn_model_3.compile(loss = 'binary_crossentropy', optimizer = opt, metrics = ['AUC'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:27:36.962682Z","iopub.execute_input":"2025-07-16T14:27:36.963275Z","iopub.status.idle":"2025-07-16T14:27:36.972135Z","shell.execute_reply.started":"2025-07-16T14:27:36.963254Z","shell.execute_reply":"2025-07-16T14:27:36.971439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nh1 = cnn_model_3.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 10,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:27:38.760149Z","iopub.execute_input":"2025-07-16T14:27:38.760782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = h1.history\nepoch_range = range(1,len(history['loss'])+1)\n\nplt.figure(figsize = [12,5])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('Loss');plt.title('Loss')\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['AUC'], label = 'Training')\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('AUC');plt.title(\"AUC\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-16T14:16:28.876022Z","iopub.status.idle":"2025-07-16T14:16:28.876262Z","shell.execute_reply.started":"2025-07-16T14:16:28.876134Z","shell.execute_reply":"2025-07-16T14:16:28.876144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt.learning_rate.assign(.0005)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nh2 = cnn_model_3.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 20,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nopt.learning_rate.assign(.0001)\nh3 = cnn_model_3.fit(\n    x = train_loader,\n    steps_per_epoch = TR_STEPS,\n    epochs = 15,\n    validation_data = val_loader,\n    validation_steps = VAL_STEPS,\n    verbose = 1\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for k in history.keys():\n    history[k]+=h2.history[k]\nfor k in history.keys():\n    history[k]+=h3.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\nplt.figure(figsize = [12,5])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label = 'Training')\nplt.plot(epoch_range, history['val_loss'], label = 'Validation')\nplt.xlabel('Epoch');plt.ylabel('Loss');plt.title(\"Loss\")\n\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['AUC'], label = 'Training')\nplt.plot(epoch_range, history['val_AUC'], label = 'Validation')\nplt.xlabel(\"Epoch\");plt.ylabel(\"AUC\");plt.title(\"AUC\")\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ncnn_model_3.save('Cancer_detection_cnn_model_3.h5')\npickle.dump(history, open(f'Cancer_Detection_model_3.pk1', 'wb'))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}