{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Load Packages"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport math\nfrom keras_preprocessing.image import ImageDataGenerator\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import *\nfrom sklearn.utils import shuffle\nimport os\nimport cv2\nimport matplotlib.patches as patches\n\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Load Data"},{"metadata":{"_cell_guid":"","_uuid":"","trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\", dtype=str)\ntest_data = pd.read_csv(\"../input/histopathologic-cancer-detection/sample_submission.csv\", dtype=str)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Some Exploratory Data Analysis"},{"metadata":{},"cell_type":"markdown","source":"### The images are labeled as 0 and 1, where 0 = No Tumor and 1 = Has Tumor"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### DataFrame of all Train Image Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_data.shape)\ntrain_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### DataFrame of all Test Image Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(test_data.shape)\ntest_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Create variables for paths to image directories, then count the number of images in each directory."},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train/\"\ntest_path = \"../input/histopathologic-cancer-detection/test/\"\n# quick look at the label stats\ntrain_data['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### See the distribution of train labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize = (6,6)) \nax = sns.countplot(train_data.label).set_title('Label Counts', fontsize = 18)\nplt.annotate(train_data.label.value_counts()[0],\n            xy = (0,train_data.label.value_counts()[0] + 2000),\n            va = 'bottom',\n            ha = 'center',\n            fontsize = 12)\nplt.annotate(train_data.label.value_counts()[1],\n            xy = (1,train_data.label.value_counts()[1] + 2000),\n            va = 'bottom',\n            ha = 'center',\n            fontsize = 12)\nplt.ylim(0,150000)\nplt.ylabel('Count', fontsize = 16)\nplt.xlabel('Labels', fontsize = 16)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Total Samples Available"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Training Images:', len(os.listdir(train_path)))\nprint('Testing Images: ', len(os.listdir(test_path)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.id = train_data.id + '.tif'\ntest_data.id = test_data.id + '.tif'\nprint(train_data.head())\nprint(test_data.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1/255, validation_split=0.20)\ntest_datagen = ImageDataGenerator(rescale=1/255)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Part 1 - Fitting the CNN to the images\n### Data set generators"},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_size = 176020\nva_size = 44005\nte_size = 57458","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"b_size = 32\n\ntr_steps = math.ceil(tr_size / b_size)\nva_steps = math.ceil(va_size / b_size)\nte_steps = math.ceil(te_size / b_size)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe = train_data,\n    directory = train_path,\n    x_col = \"id\",\n    y_col = \"label\",\n    subset = \"training\",\n    batch_size = b_size,\n    #seed = 1,\n    shuffle = True,\n    class_mode = \"categorical\",\n    target_size = (96, 96))\n\nvalid_generator = train_datagen.flow_from_dataframe(\n    dataframe = train_data,\n    directory = train_path,\n    x_col = \"id\",\n    y_col = \"label\",\n    subset = \"validation\",\n    batch_size = b_size,\n    #seed = 1,\n    shuffle = True,\n    class_mode = \"categorical\",\n    target_size = (96, 96))\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe = test_data,\n    directory = test_path,\n    x_col = \"id\",\n    y_col = None,\n    batch_size = 32,\n    seed = 1,\n    shuffle = False,\n    class_mode = None,\n    target_size = (96, 96))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plot some train images with and without cancer to visualize"},{"metadata":{"trusted":true},"cell_type":"code","source":"def training_images(seed):\n    np.random.seed(seed)\n    train_generator.reset()\n    imgs, labels = next(train_generator)\n    tr_labels = np.argmax(labels, axis=1)\n    \n    plt.figure(figsize=(10, 10))\n    for i in range(10):\n        text_class = labels[i]\n        plt.subplot(4, 5, i+1)\n        plt.imshow(imgs[i, :, :, :])\n        if(text_class[0] == 0):\n            plt.text(0, -5, 'Positive', color='r')\n        else:\n            plt.text(0, -5, 'Negative', color='b')\n        plt.axis('off')\n    plt.show()\n\ntraining_images(1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Part 2 - Building the CNN\n### Defining the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(1)\n\n# Initialising the CNN\nmodel = Sequential()\n\n# Step 1 - Convolution\nmodel.add(Cropping2D(cropping=((32, 32), (32, 32)), input_shape=(96, 96, 3)))\nmodel.add(Conv2D(32, (3, 3), padding = 'same', activation = 'relu'))\n\n# Step 2 - Pooling\nmodel.add(MaxPooling2D(2, 2))\nmodel.add(BatchNormalization())\n\n# Adding a second convolutional layer\nmodel.add(Conv2D(32, (3, 3), padding = 'same', activation = 'relu'))\nmodel.add(MaxPooling2D(2, 2))\n\n# Step 3 - Flattening\nmodel.add(Flatten())\n\n# Step 4 - Full connection\nmodel.add(Dense(64, use_bias=False))\nmodel.add(BatchNormalization())\nmodel.add(Activation(\"relu\"))\nmodel.add(Dense(2, activation = 'softmax'))\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualize the model arquitecture"},{"metadata":{"trusted":true},"cell_type":"code","source":"#from keras.utils import plot_model\n#plot_model(model, to_file='model.png')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training Routine"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time \n\nopt = keras.optimizers.Adam(learning_rate=0.00015, beta_1=0.9, beta_2=0.999, epsilon=1e-07, amsgrad=False, decay=0.0)\n\n# Compiling the CNN\nmodel.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy'])\n\nhist = model.fit_generator(train_generator, epochs=10, validation_data=valid_generator, \n                           steps_per_epoch=tr_steps, validation_steps=va_steps, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plot the results"},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs_range = range(1, len(hist.history['accuracy']) + 1)\n\nplt.figure(figsize=[12, 6])\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, hist.history['accuracy'], label='Training Accuracy')\nplt.plot(epochs_range, hist.history['val_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epoch')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, hist.history['loss'], label='Loss')\nplt.plot(epochs_range, hist.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epoch')\nplt.legend()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_pred = model.predict_generator(test_generator, steps = te_steps, verbose=1)\n\npred_classes = np.argmax(test_pred, axis=1)\n \ntest_fnames = test_generator.filenames\ntest_fnames = [x.split('.')[0] for x in test_fnames] ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame({\n    'id':test_fnames,\n    'label':pred_classes\n})\n \nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}