{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Data Description:\n*     The dataset used in the project is the Histopathologic Cancer Detection dataset rom Kaggle, the link of the dataset is: [Histopathologic Cancer Detection](https://www.kaggle.com/competitions/histopathologic-cancer-detection/overview)\n* Each input is a small pathology image that to be classified to a binary class. The positive label indicates that the center 32*32 px region of a patch contains at least one pixel of tumor tissue. \n* The train_labels.csv file provides the class of each associated image in train folder.\n* @misc{histopathologic-cancer-detection,\n    author = {Will Cukierski},\n    title = {Histopathologic Cancer Detection},\n    publisher = {Kaggle},\n    year = {2018},\n    url = {https://kaggle.com/competitions/histopathologic-cancer-detection}\n}    ","metadata":{}},{"cell_type":"code","source":"# import libaries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os, cv2\n\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import  Dropout,Flatten, Conv2D,MaxPooling2D, Dense\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:20.527887Z","iopub.execute_input":"2023-11-19T19:40:20.528353Z","iopub.status.idle":"2023-11-19T19:40:34.225981Z","shell.execute_reply.started":"2023-11-19T19:40:20.528312Z","shell.execute_reply":"2023-11-19T19:40:34.224953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read Label from kaggle\nLabel = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.227986Z","iopub.execute_input":"2023-11-19T19:40:34.228689Z","iopub.status.idle":"2023-11-19T19:40:34.606310Z","shell.execute_reply.started":"2023-11-19T19:40:34.228651Z","shell.execute_reply":"2023-11-19T19:40:34.605509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a look at the data\nprint(f'The shape of the label file is: {Label.shape}')\nprint('The first 5 rows are:')\nprint(Label.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.607451Z","iopub.execute_input":"2023-11-19T19:40:34.607750Z","iopub.status.idle":"2023-11-19T19:40:34.620981Z","shell.execute_reply.started":"2023-11-19T19:40:34.607724Z","shell.execute_reply":"2023-11-19T19:40:34.619689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The dataset has 220025 rows and 2 columns. The column id indicates the file name of the image and the label indicates the class of that image.","metadata":{}},{"cell_type":"code","source":"Label.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.623094Z","iopub.execute_input":"2023-11-19T19:40:34.623405Z","iopub.status.idle":"2023-11-19T19:40:34.667515Z","shell.execute_reply.started":"2023-11-19T19:40:34.623378Z","shell.execute_reply":"2023-11-19T19:40:34.666514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The data type in the label column is integer, need to change to object.","metadata":{}},{"cell_type":"code","source":"# change the data type of the label column from int to object\nLabel['label']=Label['label'].astype('str')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.668797Z","iopub.execute_input":"2023-11-19T19:40:34.669515Z","iopub.status.idle":"2023-11-19T19:40:34.807430Z","shell.execute_reply.started":"2023-11-19T19:40:34.669475Z","shell.execute_reply":"2023-11-19T19:40:34.806447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Label.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.809169Z","iopub.execute_input":"2023-11-19T19:40:34.809543Z","iopub.status.idle":"2023-11-19T19:40:34.859038Z","shell.execute_reply.started":"2023-11-19T19:40:34.809508Z","shell.execute_reply":"2023-11-19T19:40:34.858027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add the formate of each image to its file name for easier image loading\nLabel['id'] = Label['id'].apply(lambda x : x+ '.tif')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.860344Z","iopub.execute_input":"2023-11-19T19:40:34.860643Z","iopub.status.idle":"2023-11-19T19:40:34.949137Z","shell.execute_reply.started":"2023-11-19T19:40:34.860612Z","shell.execute_reply":"2023-11-19T19:40:34.948154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Label.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.950308Z","iopub.execute_input":"2023-11-19T19:40:34.950593Z","iopub.status.idle":"2023-11-19T19:40:34.962065Z","shell.execute_reply.started":"2023-11-19T19:40:34.950568Z","shell.execute_reply":"2023-11-19T19:40:34.961086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Label.label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:34.963354Z","iopub.execute_input":"2023-11-19T19:40:34.963848Z","iopub.status.idle":"2023-11-19T19:40:35.006728Z","shell.execute_reply.started":"2023-11-19T19:40:34.963815Z","shell.execute_reply":"2023-11-19T19:40:35.005858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Label.label.value_counts()[0]/(Label.label.value_counts()[0]+Label.label.value_counts()[1])","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:35.011635Z","iopub.execute_input":"2023-11-19T19:40:35.011970Z","iopub.status.idle":"2023-11-19T19:40:35.115697Z","shell.execute_reply.started":"2023-11-19T19:40:35.011942Z","shell.execute_reply":"2023-11-19T19:40:35.114808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There are 59% of class 0 in the dataset and 41% of class 1 in the dataset.","metadata":{}},{"cell_type":"code","source":"sns.countplot(Label,x = 'label');","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:35.116758Z","iopub.execute_input":"2023-11-19T19:40:35.117052Z","iopub.status.idle":"2023-11-19T19:40:35.571674Z","shell.execute_reply.started":"2023-11-19T19:40:35.117026Z","shell.execute_reply":"2023-11-19T19:40:35.570687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# helper function to load the image\ndef load_image(curr_dir,img_id):\n    \"\"\"function to read the image from its path \n    input: img_id is the id of the image file\n    \"\"\"\n    path = curr_dir + img_id \n    im = cv2.imread(path)\n    return im\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:35.572801Z","iopub.execute_input":"2023-11-19T19:40:35.573071Z","iopub.status.idle":"2023-11-19T19:40:35.578412Z","shell.execute_reply.started":"2023-11-19T19:40:35.573048Z","shell.execute_reply":"2023-11-19T19:40:35.577494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example image\ncurr_dir = '/kaggle/input/histopathologic-cancer-detection/train/'\ni = 5 # index of Label DataFrame\ns=load_image(curr_dir,Label.iloc[i]['id'])\nplt.imshow(s)\ncl = Label.iloc[i]['label']\nplt.title(f'Index {i} : Class {cl}')\n\n# take a look at the image\nprint(f'The size of the image is : {s.shape}')\nprint(f'The maximum value of the image is : {s.max()}')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:35.579588Z","iopub.execute_input":"2023-11-19T19:40:35.579897Z","iopub.status.idle":"2023-11-19T19:40:35.834785Z","shell.execute_reply.started":"2023-11-19T19:40:35.579872Z","shell.execute_reply":"2023-11-19T19:40:35.833750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split training dataset to train and validation sets (75%)\ntrain, validation = train_test_split(Label,test_size=0.25,random_state=401,stratify=Label.label)\nprint(train.shape,validation.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:35.835896Z","iopub.execute_input":"2023-11-19T19:40:35.836206Z","iopub.status.idle":"2023-11-19T19:40:36.294320Z","shell.execute_reply.started":"2023-11-19T19:40:35.836178Z","shell.execute_reply":"2023-11-19T19:40:36.293310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a look at the train dataset\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:36.295372Z","iopub.execute_input":"2023-11-19T19:40:36.295642Z","iopub.status.idle":"2023-11-19T19:40:36.306385Z","shell.execute_reply.started":"2023-11-19T19:40:36.295604Z","shell.execute_reply":"2023-11-19T19:40:36.305358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resest the index of train and validation dataset\ntrain.reset_index(inplace=True)\nvalidation.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:36.307938Z","iopub.execute_input":"2023-11-19T19:40:36.308341Z","iopub.status.idle":"2023-11-19T19:40:36.316077Z","shell.execute_reply.started":"2023-11-19T19:40:36.308305Z","shell.execute_reply":"2023-11-19T19:40:36.315158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a image generator \nimage_gen = ImageDataGenerator(rotation_range=30, \n                                    rescale = 1/255,\n                                    zoom_range=0.2,\n                                    horizontal_flip=True,\n                                    vertical_flip=True,\n                                    shear_range=0.1,\n                                   height_shift_range=0.1,\n                                   width_shift_range=0.1,\n                                    fill_mode='nearest')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:36.317405Z","iopub.execute_input":"2023-11-19T19:40:36.317714Z","iopub.status.idle":"2023-11-19T19:40:36.326055Z","shell.execute_reply.started":"2023-11-19T19:40:36.317688Z","shell.execute_reply":"2023-11-19T19:40:36.325158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a look at random transformed image compare with the original image\nfig,axs = plt.subplots(1,2,figsize=(6,3),layout='constrained',sharex=True,sharey=True)\naxs[0].set(title='Before',aspect=1,xticks=[],yticks=[])\naxs[0].imshow(s)\naxs[1].set(title='After',aspect=1,xticks=[],yticks=[])\naxs[1].imshow(image_gen.random_transform(np.array(s)));","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:36.327238Z","iopub.execute_input":"2023-11-19T19:40:36.327552Z","iopub.status.idle":"2023-11-19T19:40:36.604538Z","shell.execute_reply.started":"2023-11-19T19:40:36.327525Z","shell.execute_reply":"2023-11-19T19:40:36.603652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generate image for training dataset\ncurr_dir = '/kaggle/input/histopathologic-cancer-detection/train/'\ninput_size=(96,96)\nbatch_size = 32\ntrain_gen = image_gen.flow_from_dataframe(train,directory=curr_dir,x_col='id',y_col='label',\n                                          target_size=input_size,batch_size=batch_size,seed=401,\n                                          class_mode='binary',shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:40:36.605799Z","iopub.execute_input":"2023-11-19T19:40:36.606614Z","iopub.status.idle":"2023-11-19T19:47:46.021439Z","shell.execute_reply.started":"2023-11-19T19:40:36.606576Z","shell.execute_reply":"2023-11-19T19:47:46.020563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generate image for validaiton dataset\nval_gen = image_gen.flow_from_dataframe(validation,directory=curr_dir,x_col='id',y_col='label',\n                                          target_size=input_size,batch_size=batch_size,seed=401,\n                                        class_mode='binary',shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:47:46.022953Z","iopub.execute_input":"2023-11-19T19:47:46.023254Z","iopub.status.idle":"2023-11-19T19:50:06.957157Z","shell.execute_reply.started":"2023-11-19T19:47:46.023230Z","shell.execute_reply":"2023-11-19T19:50:06.956156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a CNN model\nmodel = Sequential()\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),input_shape=(96,96,3),activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),input_shape=(96,96,3),activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters=128,kernel_size=(3,3),input_shape=(96,96,3),activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),input_shape=(96,96,3),activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\nmodel.add(Flatten())\n\nmodel.add(Dense(128))\nmodel.add(Dropout(0.4))\n\nmodel.add(Dense(1,activation='sigmoid'))\n\nmodel.compile(optimizer=Adam(),loss='binary_crossentropy',\n             metrics=[tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:06.958503Z","iopub.execute_input":"2023-11-19T19:50:06.958892Z","iopub.status.idle":"2023-11-19T19:50:10.072087Z","shell.execute_reply.started":"2023-11-19T19:50:06.958858Z","shell.execute_reply":"2023-11-19T19:50:10.071184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.073317Z","iopub.execute_input":"2023-11-19T19:50:10.073607Z","iopub.status.idle":"2023-11-19T19:50:10.110982Z","shell.execute_reply.started":"2023-11-19T19:50:10.073581Z","shell.execute_reply":"2023-11-19T19:50:10.109983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_steps = len(train)/batch_size\nval_steps = len(validation)/batch_size\nprint(train_steps,val_steps)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.112305Z","iopub.execute_input":"2023-11-19T19:50:10.112596Z","iopub.status.idle":"2023-11-19T19:50:10.117822Z","shell.execute_reply.started":"2023-11-19T19:50:10.112570Z","shell.execute_reply":"2023-11-19T19:50:10.116840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model\nmodel_save = ModelCheckpoint('./Model_v1.h5',\n                            save_best_only = True,\n                            save_weights_only=True,\n                            monitor='val_loss',\n                            mode='min',verbose=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.119148Z","iopub.execute_input":"2023-11-19T19:50:10.119513Z","iopub.status.idle":"2023-11-19T19:50:10.131133Z","shell.execute_reply.started":"2023-11-19T19:50:10.119482Z","shell.execute_reply":"2023-11-19T19:50:10.130002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set an early stop to provent overfitting\nearly_stop = EarlyStopping(monitor='val_loss',min_delta=0.001,patience=4,mode='min',verbose=1,\n                          restore_best_weights= True)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.132799Z","iopub.execute_input":"2023-11-19T19:50:10.133493Z","iopub.status.idle":"2023-11-19T19:50:10.142032Z","shell.execute_reply.started":"2023-11-19T19:50:10.133460Z","shell.execute_reply":"2023-11-19T19:50:10.141154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reduce the learning rate if needed\nreduce_lr = ReduceLROnPlateau(monitor='val_loss',factor=0.3,patience=3,min_delta=0.002,\n                             mode='min',verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.143182Z","iopub.execute_input":"2023-11-19T19:50:10.143459Z","iopub.status.idle":"2023-11-19T19:50:10.153783Z","shell.execute_reply.started":"2023-11-19T19:50:10.143435Z","shell.execute_reply":"2023-11-19T19:50:10.152900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 5\nresults = model.fit(train_gen,\n                    steps_per_epoch=train_steps,\n                    epochs=epochs,\n                    validation_data=val_gen,\n                    validation_steps=val_steps,\n                   callbacks=[model_save,early_stop,reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:50:10.154852Z","iopub.execute_input":"2023-11-19T19:50:10.155129Z","iopub.status.idle":"2023-11-19T21:34:55.177816Z","shell.execute_reply.started":"2023-11-19T19:50:10.155087Z","shell.execute_reply":"2023-11-19T21:34:55.177014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = pd.DataFrame(model.history.history)\nhistory.columns=['loss','auc','val_loss','val_auc','lr']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T21:34:55.182972Z","iopub.execute_input":"2023-11-19T21:34:55.183272Z","iopub.status.idle":"2023-11-19T21:34:55.188398Z","shell.execute_reply.started":"2023-11-19T21:34:55.183246Z","shell.execute_reply":"2023-11-19T21:34:55.187507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot diagnostic learning curves\ndef summarize_diagnostics(history):\n    # plot loss\n    plt.figure(figsize=(8,8))\n    plt.subplot(211)\n    plt.title('Cross Entropy Loss')\n    plt.plot(history['loss'], color='blue', label='train')\n    plt.plot(history['val_loss'], color='orange', label='test')\n    plt.legend()\n    # plot auc\n    plt.subplot(212)\n    plt.title('Classification Accuracy')\n    plt.plot(history['auc'], color='blue', label='train')\n    plt.plot(history['val_auc'], color='orange', label='test')\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T21:34:55.189753Z","iopub.execute_input":"2023-11-19T21:34:55.190074Z","iopub.status.idle":"2023-11-19T21:34:55.198460Z","shell.execute_reply.started":"2023-11-19T21:34:55.190043Z","shell.execute_reply":"2023-11-19T21:34:55.197710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summarize_diagnostics(history)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T21:34:55.199481Z","iopub.execute_input":"2023-11-19T21:34:55.199772Z","iopub.status.idle":"2023-11-19T21:34:55.807008Z","shell.execute_reply.started":"2023-11-19T21:34:55.199747Z","shell.execute_reply":"2023-11-19T21:34:55.806097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss=pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\npreds=[]\n\nfor image_id in ss['id']:\n    image = Image.open('/kaggle/input/histopathologic-cancer-detection/test/'+ image_id+'.tif')\n    image = image.resize((96,96))\n    image = np.expand_dims(image,axis=0)\n    preds.append(np.ravel(np.round(model.predict(image))))\n\nss['label'] = preds\n\nss.to_csv('results.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-19T21:34:55.808387Z","iopub.execute_input":"2023-11-19T21:34:55.808762Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\npkl_filename = 'Model_v2.pkl'\n\nwith open(pkl_filename,'wb') as file:\n    pickle.dump(model,file)\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}