{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import useful libraries for EDA and importation of data\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T11:51:07.378517Z","iopub.execute_input":"2022-07-31T11:51:07.379038Z","iopub.status.idle":"2022-07-31T11:51:07.412093Z","shell.execute_reply.started":"2022-07-31T11:51:07.37895Z","shell.execute_reply":"2022-07-31T11:51:07.410962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\nprint('Training Shape:',train_labels.shape)\nprint('Input example :')\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:08.188055Z","iopub.execute_input":"2022-07-31T11:51:08.191187Z","iopub.status.idle":"2022-07-31T11:51:08.707239Z","shell.execute_reply.started":"2022-07-31T11:51:08.191137Z","shell.execute_reply":"2022-07-31T11:51:08.705851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_labels['label'] = train_labels['label'].astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:08.71Z","iopub.execute_input":"2022-07-31T11:51:08.71087Z","iopub.status.idle":"2022-07-31T11:51:08.757598Z","shell.execute_reply.started":"2022-07-31T11:51:08.710823Z","shell.execute_reply":"2022-07-31T11:51:08.756317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Look at Test/Train size difference\n\nprint('Training samples: ',len(os.listdir('../input/histopathologic-cancer-detection/train/')))\nprint('Test samples: ',len(os.listdir('../input/histopathologic-cancer-detection/test/')))\n\nTrainProportion = len(os.listdir('../input/histopathologic-cancer-detection/train/'))/ (len(os.listdir('../input/histopathologic-cancer-detection/train/')) + len(os.listdir('../input/histopathologic-cancer-detection/test/')))\n\nprint('Proportion of data in Train set: ',TrainProportion)\nprint('Proportion of data in Test set: ',1- TrainProportion)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:09.094859Z","iopub.execute_input":"2022-07-31T11:51:09.095262Z","iopub.status.idle":"2022-07-31T11:51:25.347836Z","shell.execute_reply.started":"2022-07-31T11:51:09.095229Z","shell.execute_reply":"2022-07-31T11:51:25.346313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Establish distribution of training labels\n\ntrain_labels['label'].value_counts()\ntrain_labels['label'].value_counts().plot(kind='bar',grid=True,title='Distribution of Binary Classes')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:25.350447Z","iopub.execute_input":"2022-07-31T11:51:25.352089Z","iopub.status.idle":"2022-07-31T11:51:25.613183Z","shell.execute_reply.started":"2022-07-31T11:51:25.352029Z","shell.execute_reply":"2022-07-31T11:51:25.611853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make occurence of 0s equivalent to 1s in the training data\n\ntrain_pos = train_labels[train_labels['label'] == 1]\ntrain_neg = train_labels[train_labels['label'] == 0]\ntrain_neg = train_neg.sample(n = len(train_pos))\n\n# Check new sizes\nprint(len(train_neg))\nprint(len(train_pos))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:25.614615Z","iopub.execute_input":"2022-07-31T11:51:25.615059Z","iopub.status.idle":"2022-07-31T11:51:25.651305Z","shell.execute_reply.started":"2022-07-31T11:51:25.615028Z","shell.execute_reply":"2022-07-31T11:51:25.650081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concat two datasets, confirm new set is balance w/ Pie chart\n\ntrain_bal = pd.concat([train_neg,train_pos]).sample(frac=1, random_state=42).reset_index(drop=True)\ntrain_bal.head()\n\ntrain_bal['label'].value_counts().plot(kind='pie',title='Balnced Data')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:25.654971Z","iopub.execute_input":"2022-07-31T11:51:25.65549Z","iopub.status.idle":"2022-07-31T11:51:25.800546Z","shell.execute_reply.started":"2022-07-31T11:51:25.655448Z","shell.execute_reply":"2022-07-31T11:51:25.79899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimg\nimport matplotlib.pyplot as plt\n\nimg = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{train_bal.iloc[1,0]}.tif')\nimgplot = plt.imshow(img)\n\nprint('Shape of image:', img.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:25.802746Z","iopub.execute_input":"2022-07-31T11:51:25.803217Z","iopub.status.idle":"2022-07-31T11:51:26.051147Z","shell.execute_reply.started":"2022-07-31T11:51:25.80316Z","shell.execute_reply":"2022-07-31T11:51:26.049887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rndm_img = np.random.choice(train_bal.index,64)\nfig, ax = plt.subplots(8, 8,figsize=(14,14))\n\nfor i in range(0, rndm_img.shape[0]):\n    ax = plt.subplot(8, 8, i+1)\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{train_bal.iloc[rndm_img[i],0]}.tif')\n    ax.imshow(img)\n    label = train_bal.iloc[rndm_img[i],1]\n    ax.set_title('Label: %s'%label)\n    \nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:26.052563Z","iopub.execute_input":"2022-07-31T11:51:26.053049Z","iopub.status.idle":"2022-07-31T11:51:34.451075Z","shell.execute_reply.started":"2022-07-31T11:51:26.053003Z","shell.execute_reply":"2022-07-31T11:51:34.449548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the Data\nfrom sklearn.model_selection import train_test_split\ntrain_df, val_df = train_test_split(train_bal, test_size=0.2, random_state=42, \n                                      stratify=train_bal.label)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:34.452801Z","iopub.execute_input":"2022-07-31T11:51:34.454105Z","iopub.status.idle":"2022-07-31T11:51:35.175363Z","shell.execute_reply.started":"2022-07-31T11:51:34.454064Z","shell.execute_reply":"2022-07-31T11:51:35.174081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create labels for model\n\n# Train\ntrain_df['id'] = train_df['id']+'.tif'\ntrain_df['label'] = train_df['label'].astype(str)\nprint(train_df.shape)\n# Test\nval_df['id'] = val_df['id']+'.tif'\nval_df['label'] = val_df['label'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:35.177617Z","iopub.execute_input":"2022-07-31T11:51:35.178064Z","iopub.status.idle":"2022-07-31T11:51:35.33807Z","shell.execute_reply.started":"2022-07-31T11:51:35.17802Z","shell.execute_reply":"2022-07-31T11:51:35.336608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\npreprocess_input = tf.keras.applications.mobilenet_v2.preprocess_input\nrescale = tf.keras.layers.Rescaling(1./127.5, offset=-1)\nIMG_SHAPE = (96, 96, 3)\nbase_model = tf.keras.applications.MobileNetV2(input_shape=IMG_SHAPE,\n                                               include_top=False,\n                                               weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:35.340283Z","iopub.execute_input":"2022-07-31T11:51:35.340835Z","iopub.status.idle":"2022-07-31T11:51:47.641354Z","shell.execute_reply.started":"2022-07-31T11:51:35.340775Z","shell.execute_reply":"2022-07-31T11:51:47.639885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen=ImageDataGenerator(rescale=1./127.5,dtype=tf.float32) # To adjust pixel contents between -1/1\ntrain_generator=train_datagen.flow_from_dataframe(dataframe=train_df, directory=\"../input/histopathologic-cancer-detection/train/\",\n                x_col=\"id\",y_col=\"label\",batch_size=64,seed=42,shuffle=True,\n                class_mode=\"binary\",target_size=(96,96))\nvalid_generator=train_datagen.flow_from_dataframe(dataframe=val_df, directory=\"../input/histopathologic-cancer-detection/train/\",\n                x_col=\"id\",y_col=\"label\",batch_size=64,seed=42,shuffle=True,\n                class_mode=\"binary\",target_size=(96,96))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T11:51:47.64829Z","iopub.execute_input":"2022-07-31T11:51:47.64869Z","iopub.status.idle":"2022-07-31T12:01:41.185658Z","shell.execute_reply.started":"2022-07-31T11:51:47.648658Z","shell.execute_reply":"2022-07-31T12:01:41.184168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable = False\nbase_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.187304Z","iopub.execute_input":"2022-07-31T12:01:41.187958Z","iopub.status.idle":"2022-07-31T12:01:41.244056Z","shell.execute_reply.started":"2022-07-31T12:01:41.187926Z","shell.execute_reply":"2022-07-31T12:01:41.242413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_average_layer = tf.keras.layers.GlobalAveragePooling2D()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.245948Z","iopub.execute_input":"2022-07-31T12:01:41.248195Z","iopub.status.idle":"2022-07-31T12:01:41.255616Z","shell.execute_reply.started":"2022-07-31T12:01:41.248151Z","shell.execute_reply":"2022-07-31T12:01:41.254113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_layer = tf.keras.layers.Dense(1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.25737Z","iopub.execute_input":"2022-07-31T12:01:41.259047Z","iopub.status.idle":"2022-07-31T12:01:41.26997Z","shell.execute_reply.started":"2022-07-31T12:01:41.259004Z","shell.execute_reply":"2022-07-31T12:01:41.268404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=IMG_SHAPE,dtype = tf.uint8)\nx = tf.cast(inputs,tf.float32)\nx = preprocess_input(x)\nx = base_model(x, training=False)\nx = global_average_layer(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutputs = prediction_layer(x)\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.272028Z","iopub.execute_input":"2022-07-31T12:01:41.273206Z","iopub.status.idle":"2022-07-31T12:01:41.765485Z","shell.execute_reply.started":"2022-07-31T12:01:41.273147Z","shell.execute_reply":"2022-07-31T12:01:41.764161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_learning_rate = 0.0001\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=base_learning_rate),\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.767336Z","iopub.execute_input":"2022-07-31T12:01:41.76797Z","iopub.status.idle":"2022-07-31T12:01:41.790749Z","shell.execute_reply.started":"2022-07-31T12:01:41.767879Z","shell.execute_reply":"2022-07-31T12:01:41.78946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.794632Z","iopub.execute_input":"2022-07-31T12:01:41.794972Z","iopub.status.idle":"2022-07-31T12:01:41.815605Z","shell.execute_reply.started":"2022-07-31T12:01:41.794941Z","shell.execute_reply":"2022-07-31T12:01:41.814329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.trainable_variables)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.81752Z","iopub.execute_input":"2022-07-31T12:01:41.818436Z","iopub.status.idle":"2022-07-31T12:01:41.827372Z","shell.execute_reply.started":"2022-07-31T12:01:41.818382Z","shell.execute_reply":"2022-07-31T12:01:41.826002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_step=train_generator.n//train_generator.batch_size\nval_step=valid_generator.n//valid_generator.batch_size\n\nprint(train_step)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.831283Z","iopub.execute_input":"2022-07-31T12:01:41.831653Z","iopub.status.idle":"2022-07-31T12:01:41.840614Z","shell.execute_reply.started":"2022-07-31T12:01:41.831624Z","shell.execute_reply":"2022-07-31T12:01:41.838881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_model= model.fit(train_generator,\n                    steps_per_epoch=train_step,\n                    validation_data=valid_generator,\n                    validation_steps=val_step,\n                    epochs=15, verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:01:41.842722Z","iopub.execute_input":"2022-07-31T12:01:41.844217Z","iopub.status.idle":"2022-07-31T13:45:34.463812Z","shell.execute_reply.started":"2022-07-31T12:01:41.844173Z","shell.execute_reply":"2022-07-31T13:45:34.462509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history_model.history['accuracy']\nval_acc = history_model.history['val_accuracy']\n\nloss = history_model.history['loss']\nval_loss = history_model.history['val_loss']\n\nplt.figure(figsize=(8, 8))\nplt.subplot(2, 1, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.ylabel('Accuracy')\nplt.ylim([min(plt.ylim()),1])\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(2, 1, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.ylabel('Cross Entropy')\nplt.ylim([0,1.0])\nplt.title('Training and Validation Loss')\nplt.xlabel('epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:45:34.465489Z","iopub.execute_input":"2022-07-31T13:45:34.466043Z","iopub.status.idle":"2022-07-31T13:45:34.865048Z","shell.execute_reply.started":"2022-07-31T13:45:34.465981Z","shell.execute_reply":"2022-07-31T13:45:34.863824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Data - prepare generator\n\ntest_set = os.listdir('../input/histopathologic-cancer-detection/test/')\ntest_df = pd.DataFrame(test_set)\ntest_df.columns = ['id']\ntest_df.head()\n\ntest_datagen=ImageDataGenerator(rescale=1./127.5,dtype=tf.float32)\n\ntest_generator=test_datagen.flow_from_dataframe(dataframe=test_df,directory=\"../input/histopathologic-cancer-detection/test/\",\n                x_col=\"id\",batch_size=64,seed=42,shuffle=False,\n                class_mode=None,target_size=(96,96))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:45:34.867215Z","iopub.execute_input":"2022-07-31T13:45:34.86778Z","iopub.status.idle":"2022-07-31T13:49:02.160223Z","shell.execute_reply.started":"2022-07-31T13:45:34.867702Z","shell.execute_reply":"2022-07-31T13:49:02.158847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST=test_generator.n\nmodelpreds = base_model.predict(test_generator,steps=STEP_SIZE_TEST, verbose = 1)\n\npredictions = modelpreds\nprint (len(predictions))\nfor pred in modelpreds:\n    if pred.any() >= 0.5:\n        predictions.append(1)\n    else:\n        predictions.append(0)\nsubmission = test_df.copy()\nsubmission['id']=submission['id'].str[:-4] # Remove \".tif\"\nsubmission['label']=predictions\nsubmission.head()\n\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:34:09.389541Z","iopub.execute_input":"2022-07-31T14:34:09.390097Z","iopub.status.idle":"2022-07-31T14:38:17.920577Z","shell.execute_reply.started":"2022-07-31T14:34:09.390052Z","shell.execute_reply":"2022-07-31T14:38:17.918633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('../output/model')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:45:03.304588Z","iopub.execute_input":"2022-07-31T14:45:03.305146Z","iopub.status.idle":"2022-07-31T14:45:40.852436Z","shell.execute_reply.started":"2022-07-31T14:45:03.30505Z","shell.execute_reply":"2022-07-31T14:45:40.85107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}