{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport shutil\nprint(os.listdir(\"../input/histopathologic-cancer-detection\"))\n\nfrom glob import glob \nfrom skimage.io import imread\nimport gc\n\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-17T10:43:45.013187Z","iopub.execute_input":"2021-07-17T10:43:45.013514Z","iopub.status.idle":"2021-07-17T10:43:45.1435Z","shell.execute_reply.started":"2021-07-17T10:43:45.013484Z","shell.execute_reply":"2021-07-17T10:43:45.142654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading dataset","metadata":{}},{"cell_type":"code","source":"base_tile_dir = '../input/histopathologic-cancer-detection/train/'\ndf = pd.DataFrame({'path': glob(os.path.join(base_tile_dir,'*.tif'))})\ndf['id'] = df.path.map(lambda x: x.split('/')[4].split(\".\")[0])\nlabels = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ndf_data = df.merge(labels, on = \"id\")\n\n# removing this image because it caused a training error previously\ndf_data = df_data[df_data['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n\n# removing this image because it's black\ndf_data = df_data[df_data['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\ndf_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:42:53.398131Z","iopub.execute_input":"2021-07-17T10:42:53.398494Z","iopub.status.idle":"2021-07-17T10:42:58.954886Z","shell.execute_reply.started":"2021-07-17T10:42:53.398461Z","shell.execute_reply":"2021-07-17T10:42:58.954049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"label\",data=df_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:44:06.49819Z","iopub.execute_input":"2021-07-17T10:44:06.498504Z","iopub.status.idle":"2021-07-17T10:44:06.649925Z","shell.execute_reply.started":"2021-07-17T10:44:06.498474Z","shell.execute_reply":"2021-07-17T10:44:06.64895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Split X and y in train/test and build folders","metadata":{}},{"cell_type":"code","source":"SAMPLE_SIZE = 80000 # load 80k negative examples\n\n# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df_data[df_data['label'] == 0].sample(SAMPLE_SIZE, random_state = 101)\n# filter out class 1\ndf_1 = df_data[df_data['label'] == 1].sample(SAMPLE_SIZE, random_state = 101)\n\n# concat the dataframes\ndf_data = shuffle(pd.concat([df_0, df_1], axis=0).reset_index(drop=True))\n\n# train_test_split # stratify=y creates a balanced validation set.\ny = df_data['label']\ndf_train, df_val = train_test_split(df_data, test_size=0.20, random_state=101, stratify=y)\n\n# Create directories\ntrain_path = 'base_dir/train'\nvalid_path = 'base_dir/valid'\ntest_path = '../input/histopathologic-cancer-detection/test'\nfor fold in [train_path, valid_path]:\n    for subf in [\"0\", \"1\"]:\n        os.makedirs(os.path.join(fold, subf))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:44:47.089035Z","iopub.execute_input":"2021-07-17T10:44:47.089463Z","iopub.status.idle":"2021-07-17T10:44:47.469493Z","shell.execute_reply.started":"2021-07-17T10:44:47.089419Z","shell.execute_reply":"2021-07-17T10:44:47.468467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"label\",data=df_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:45:02.608748Z","iopub.execute_input":"2021-07-17T10:45:02.609069Z","iopub.status.idle":"2021-07-17T10:45:02.737055Z","shell.execute_reply.started":"2021-07-17T10:45:02.60904Z","shell.execute_reply":"2021-07-17T10:45:02.73601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the id as the index in df_data\ndf_data.set_index('id', inplace=True)\ndf_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:48:50.258727Z","iopub.execute_input":"2021-07-17T10:48:50.259085Z","iopub.status.idle":"2021-07-17T10:48:50.289306Z","shell.execute_reply.started":"2021-07-17T10:48:50.259054Z","shell.execute_reply":"2021-07-17T10:48:50.288431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:51:47.253013Z","iopub.execute_input":"2021-07-17T10:51:47.253439Z","iopub.status.idle":"2021-07-17T10:51:47.257851Z","shell.execute_reply.started":"2021-07-17T10:51:47.253403Z","shell.execute_reply":"2021-07-17T10:51:47.257013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(5)):\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:51:57.038146Z","iopub.execute_input":"2021-07-17T10:51:57.038494Z","iopub.status.idle":"2021-07-17T10:51:57.08099Z","shell.execute_reply.started":"2021-07-17T10:51:57.038464Z","shell.execute_reply":"2021-07-17T10:51:57.07985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for image in tqdm(df_train['id'].values):\n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    label = str(df_data.loc[image,'label']) # get the label for a certain image\n    src = os.path.join('../input/histopathologic-cancer-detection/train', fname)\n    dst = os.path.join(train_path, label, fname)\n    shutil.copyfile(src, dst)\n\nfor image in tqdm(df_val['id'].values):\n    fname = image + '.tif'\n    label = str(df_data.loc[image,'label']) # get the label for a certain image\n    src = os.path.join('../input/histopathologic-cancer-detection/train', fname)\n    dst = os.path.join(valid_path, label, fname)\n    shutil.copyfile(src, dst)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T10:52:30.978131Z","iopub.execute_input":"2021-07-17T10:52:30.978456Z","iopub.status.idle":"2021-07-17T11:09:14.644277Z","shell.execute_reply.started":"2021-07-17T10:52:30.978428Z","shell.execute_reply":"2021-07-17T11:09:14.643386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '../input/histopathologic-cancer-detection/test'","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:17:57.7653Z","iopub.execute_input":"2021-07-17T11:17:57.765698Z","iopub.status.idle":"2021-07-17T11:17:57.773724Z","shell.execute_reply.started":"2021-07-17T11:17:57.765647Z","shell.execute_reply":"2021-07-17T11:17:57.771103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\nIMAGE_SIZE = 96\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 32\nval_batch_size = 32\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n\ndatagen = ImageDataGenerator(preprocessing_function=lambda x:(x - x.mean()) / x.std() if x.std() > 0 else x,\n                            horizontal_flip=True,\n                            vertical_flip=True)\n\ntrain_gen = datagen.flow_from_directory(train_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=train_batch_size,\n                                        class_mode='binary')\n\nval_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=val_batch_size,\n                                        class_mode='binary')\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=1,\n                                        class_mode='binary',\n                                        shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:22:37.557841Z","iopub.execute_input":"2021-07-17T11:22:37.55821Z","iopub.status.idle":"2021-07-17T11:22:47.728301Z","shell.execute_reply.started":"2021-07-17T11:22:37.558169Z","shell.execute_reply":"2021-07-17T11:22:47.727354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define the model¶\n**Model structure (optimizer: Adam):**\n\n* In\n* [Conv2D*3 -> MaxPool2D -> Dropout] x3 --> (filters = 16, 32, 64)\n* Flatten\n* Dense (256)\n* Dropout\n* Out","metadata":{}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization, Activation\nfrom keras.layers import Conv2D, MaxPool2D\nfrom keras.optimizers import RMSprop, Adam\n\nkernel_size = (3,3)\npool_size= (2,2)\nfirst_filters = 32\nsecond_filters = 64\nthird_filters = 128\n\ndropout_conv = 0.3\ndropout_dense = 0.5\n\nmodel = Sequential()\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu', input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3)))\nmodel.add(Conv2D(first_filters, kernel_size, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(MaxPool2D(pool_size = pool_size)) \nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(second_filters, kernel_size, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(Conv2D(second_filters, kernel_size, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(MaxPool2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(third_filters, kernel_size, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(Conv2D(third_filters, kernel_size, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(MaxPool2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\n#model.add(GlobalAveragePooling2D())\nmodel.add(Flatten())\nmodel.add(Dense(256, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(Dropout(dropout_dense))\nmodel.add(Dense(1, activation = \"sigmoid\"))\n\n# Compile the model\nmodel.compile(Adam(0.01), loss = \"binary_crossentropy\", metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:23:30.223138Z","iopub.execute_input":"2021-07-17T11:23:30.223467Z","iopub.status.idle":"2021-07-17T11:23:32.494521Z","shell.execute_reply.started":"2021-07-17T11:23:30.223434Z","shell.execute_reply":"2021-07-17T11:23:32.493687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:25:34.924075Z","iopub.execute_input":"2021-07-17T11:25:34.924819Z","iopub.status.idle":"2021-07-17T11:25:34.956097Z","shell.execute_reply.started":"2021-07-17T11:25:34.924741Z","shell.execute_reply":"2021-07-17T11:25:34.953989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ReduceLROnPlateau\nearlystopper = EarlyStopping(monitor='val_loss', patience=2, verbose=1, restore_best_weights=True)\nreducel = ReduceLROnPlateau(monitor='val_loss', patience=1, verbose=1, factor=0.1)\nhistory = model.fit(train_gen, steps_per_epoch=train_steps, \n                    validation_data=val_gen,\n                    validation_steps=val_steps,\n                    epochs=13,\n                   callbacks=[reducel, earlystopper])","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:26:36.784082Z","iopub.execute_input":"2021-07-17T11:26:36.784453Z","iopub.status.idle":"2021-07-17T11:48:16.099989Z","shell.execute_reply.started":"2021-07-17T11:26:36.784421Z","shell.execute_reply":"2021-07-17T11:48:16.099139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history.history['val_accuracy']\n\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:39:34.396003Z","iopub.execute_input":"2021-07-17T12:39:34.396349Z","iopub.status.idle":"2021-07-17T12:39:34.539354Z","shell.execute_reply.started":"2021-07-17T12:39:34.396319Z","shell.execute_reply":"2021-07-17T12:39:34.538535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, roc_auc_score\nimport matplotlib.pyplot as plt\n\n# make a prediction\ny_pred_keras = model.predict(test_gen, steps=len(df_val), verbose=1)\nfpr_keras, tpr_keras, thresholds_keras = roc_curve(test_gen.classes, y_pred_keras)\nauc_keras = auc(fpr_keras, tpr_keras)\nauc_keras","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:50:53.933076Z","iopub.execute_input":"2021-07-17T11:50:53.933423Z","iopub.status.idle":"2021-07-17T11:52:24.802931Z","shell.execute_reply.started":"2021-07-17T11:50:53.933389Z","shell.execute_reply":"2021-07-17T11:52:24.802084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_keras, tpr_keras, label='area = {:.3f}'.format(auc_keras))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:52:48.177698Z","iopub.execute_input":"2021-07-17T11:52:48.178026Z","iopub.status.idle":"2021-07-17T11:52:48.33583Z","shell.execute_reply.started":"2021-07-17T11:52:48.177996Z","shell.execute_reply":"2021-07-17T11:52:48.33478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_test_dir = '../input/histopathologic-cancer-detection/test/'\ntest_files = glob(os.path.join(base_test_dir,'*.tif'))\nsubmission = pd.DataFrame()\nfile_batch = 5000\nmax_idx = len(test_files)\nfor idx in range(0, max_idx, file_batch):\n    print(\"Indexes: %i - %i\"%(idx, idx+file_batch))\n    test_df = pd.DataFrame({'path': test_files[idx:idx+file_batch]})\n    test_df['id'] = test_df.path.map(lambda x: x.split('/')[4].split(\".\")[0])\n    test_df['image'] = test_df['path'].map(imread)\n    K_test = np.stack(test_df[\"image\"].values)\n    K_test = (K_test - K_test.mean()) / K_test.std()\n    predictions = model.predict(K_test)\n    test_df['label'] = predictions\n    submission = pd.concat([submission, test_df[[\"id\", \"label\"]]])\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:02:04.663568Z","iopub.execute_input":"2021-07-17T12:02:04.663939Z","iopub.status.idle":"2021-07-17T12:03:54.11579Z","shell.execute_reply.started":"2021-07-17T12:02:04.663909Z","shell.execute_reply":"2021-07-17T12:03:54.11478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_test = imread(test_files[10], plugin='matplotlib')\nplt.imshow(img_test)\nplt.show()\n\nimg_test_norm=(img_test - img_test.mean()) / img_test.std()\nval=np.expand_dims(img_test_norm, axis=0)\npredictions_test = model.predict(val)\nprint(predictions_test[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:16:43.042914Z","iopub.execute_input":"2021-07-17T12:16:43.04329Z","iopub.status.idle":"2021-07-17T12:16:43.23585Z","shell.execute_reply.started":"2021-07-17T12:16:43.04326Z","shell.execute_reply":"2021-07-17T12:16:43.234799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['label']=np.where(submission['label'] > 0.5, 1,0)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:27:09.903311Z","iopub.execute_input":"2021-07-17T12:27:09.903655Z","iopub.status.idle":"2021-07-17T12:27:09.909423Z","shell.execute_reply.started":"2021-07-17T12:27:09.903607Z","shell.execute_reply":"2021-07-17T12:27:09.908434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission\n# Delete the test_dir directory we created to prevent a Kaggle error.\n# Kaggle allows a max of 500 files to be saved.\n\n# shutil.rmtree(train_path)\n# shutil.rmtree(valid_path)\nsubmission.to_csv(\"submission_new.csv\", index = False, header = True)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:27:17.03878Z","iopub.execute_input":"2021-07-17T12:27:17.039115Z","iopub.status.idle":"2021-07-17T12:27:17.203709Z","shell.execute_reply.started":"2021-07-17T12:27:17.039086Z","shell.execute_reply":"2021-07-17T12:27:17.202746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"submission_new.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:27:19.747018Z","iopub.execute_input":"2021-07-17T12:27:19.747322Z","iopub.status.idle":"2021-07-17T12:27:19.799204Z","shell.execute_reply.started":"2021-07-17T12:27:19.747296Z","shell.execute_reply":"2021-07-17T12:27:19.798196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:19:51.172967Z","iopub.execute_input":"2021-07-17T12:19:51.17336Z","iopub.status.idle":"2021-07-17T12:19:51.298832Z","shell.execute_reply.started":"2021-07-17T12:19:51.173325Z","shell.execute_reply":"2021-07-17T12:19:51.298001Z"},"trusted":true},"execution_count":null,"outputs":[]}]}