{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Histopathologic Cancer Detection**","metadata":{}},{"cell_type":"code","source":"# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport matplotlib.pyplot as plt\n\n# Any results you write to the current directory are saved as output.\n\nfrom time import time\nimport seaborn as sns\nimport plotly.graph_objects as go\n\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\n\n\n# import the necessary packages\nfrom keras.models import Sequential\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras.optimizers import Adam\nfrom keras import backend as K\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-10-24T12:55:43.372555Z","iopub.execute_input":"2023-10-24T12:55:43.373587Z","iopub.status.idle":"2023-10-24T12:55:53.034122Z","shell.execute_reply.started":"2023-10-24T12:55:43.373557Z","shell.execute_reply":"2023-10-24T12:55:53.033322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv',dtype=str)\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.035381Z","iopub.execute_input":"2023-10-24T12:55:53.036485Z","iopub.status.idle":"2023-10-24T12:55:53.449544Z","shell.execute_reply.started":"2023-10-24T12:55:53.036457Z","shell.execute_reply":"2023-10-24T12:55:53.448928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Descriptive Analytics for given Dataset\n\nprint(df.label.value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.450369Z","iopub.execute_input":"2023-10-24T12:55:53.451282Z","iopub.status.idle":"2023-10-24T12:55:53.468630Z","shell.execute_reply.started":"2023-10-24T12:55:53.451259Z","shell.execute_reply":"2023-10-24T12:55:53.467693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#add .tif to ids in the dataframe to use flow_from_dataframe\ndf[\"id\"]=df[\"id\"].apply(lambda x : x +\".tif\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.470835Z","iopub.execute_input":"2023-10-24T12:55:53.471218Z","iopub.status.idle":"2023-10-24T12:55:53.544734Z","shell.execute_reply.started":"2023-10-24T12:55:53.471183Z","shell.execute_reply":"2023-10-24T12:55:53.543851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/histopathologic-cancer-detection/train/'\nvalid_path = '/kaggle/input/histopathologic-cancer-detection/test/'","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.545784Z","iopub.execute_input":"2023-10-24T12:55:53.546092Z","iopub.status.idle":"2023-10-24T12:55:53.550134Z","shell.execute_reply.started":"2023-10-24T12:55:53.546067Z","shell.execute_reply":"2023-10-24T12:55:53.549389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(validation_split=0.20,\n                          rescale=1/255.0)","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.551316Z","iopub.execute_input":"2023-10-24T12:55:53.551907Z","iopub.status.idle":"2023-10-24T12:55:53.563265Z","shell.execute_reply.started":"2023-10-24T12:55:53.551858Z","shell.execute_reply":"2023-10-24T12:55:53.561727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator=train_datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=train_path,\n    x_col=\"id\",\n    y_col=\"label\",\n    subset=\"training\",\n    batch_size=64,\n    shuffle=True,\n    class_mode=\"binary\",\n    target_size=(96,96))","metadata":{"execution":{"iopub.status.busy":"2023-10-24T12:55:53.564789Z","iopub.execute_input":"2023-10-24T12:55:53.565116Z","iopub.status.idle":"2023-10-24T13:10:41.724463Z","shell.execute_reply.started":"2023-10-24T12:55:53.565089Z","shell.execute_reply":"2023-10-24T13:10:41.723434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_generator=train_datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=valid_path,\n    x_col=\"id\",\n    y_col=\"label\",\n    subset=\"validation\",\n    batch_size=64,\n    shuffle=True,\n    class_mode=\"binary\",\n    target_size=(96,96))\n","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:10:41.725761Z","iopub.execute_input":"2023-10-24T13:10:41.726173Z","iopub.status.idle":"2023-10-24T13:13:29.875189Z","shell.execute_reply.started":"2023-10-24T13:10:41.726138Z","shell.execute_reply":"2023-10-24T13:13:29.874208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nfor i in range(12):\n    img, label = train_generator.next()\n    plt.subplot(3,4,i+1)\n    plt.imshow(img[0])\n    plt.title(\"Label = {}\".format(label[0]))","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:13:29.876446Z","iopub.execute_input":"2023-10-24T13:13:29.876962Z","iopub.status.idle":"2023-10-24T13:13:38.479358Z","shell.execute_reply.started":"2023-10-24T13:13:29.876933Z","shell.execute_reply":"2023-10-24T13:13:38.478123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define the model","metadata":{}},{"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), padding = 'same', activation = 'relu', input_shape = (96, 96, 3)))\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(MaxPooling2D(pool_size=(3,3)))\n \nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:13:38.481775Z","iopub.execute_input":"2023-10-24T13:13:38.482029Z","iopub.status.idle":"2023-10-24T13:13:38.727685Z","shell.execute_reply.started":"2023-10-24T13:13:38.481984Z","shell.execute_reply":"2023-10-24T13:13:38.726122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam' , loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:13:38.729636Z","iopub.execute_input":"2023-10-24T13:13:38.730033Z","iopub.status.idle":"2023-10-24T13:13:38.747156Z","shell.execute_reply.started":"2023-10-24T13:13:38.729980Z","shell.execute_reply":"2023-10-24T13:13:38.746012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nEPOCHS=20","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:13:38.748751Z","iopub.execute_input":"2023-10-24T13:13:38.749142Z","iopub.status.idle":"2023-10-24T13:13:38.753516Z","shell.execute_reply.started":"2023-10-24T13:13:38.749109Z","shell.execute_reply":"2023-10-24T13:13:38.752757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nearlystopper = EarlyStopping(monitor='val_accuracy', patience=3, verbose=1, restore_best_weights=True)\nreducel = ReduceLROnPlateau(monitor='val_accuracy', patience=2, verbose=1, factor=0.1)\n\n\nhistory = model.fit_generator(generator=train_generator, \n                    steps_per_epoch=STEP_SIZE_TRAIN, \n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=EPOCHS,\n                   callbacks=[reducel, earlystopper])\n","metadata":{"execution":{"iopub.status.busy":"2023-10-24T13:13:38.754643Z","iopub.execute_input":"2023-10-24T13:13:38.756364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model Evaluation","metadata":{}},{"cell_type":"code","source":"def show_final_history(history):\n    fig, ax = plt.subplots(1, 2, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('accuracy')\n    ax[1].plot(history.epoch, history.history[\"accuracy\"], label=\"Train acc\")\n    ax[1].plot(history.epoch, history.history[\"val_accuracy\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_final_history(history)\nprint(\"Validation Accuracy: \" + str(history.history['val_accuracy'][-1:]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}