{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Histopathologic Cancer Detection"},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom skimage.io import imread\nimport keras.backend as k\nimport tensorflow as tf\nimport os\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelBinarizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame({'path': glob(os.path.join('../input/train', '*.tif'))})\ndf['id'] = df.path.map(lambda x: x.split('/')[3].split(\".\")[0])\nlabels = pd.read_csv('../input/train_labels.csv')\ndf = df.merge(labels, on=\"id\")\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df0 = df[df.label == 0].sample(500, random_state=42)\ndf1 = df[df.label == 1].sample(500, random_state=42)\ndf = pd.concat([df0, df1], ignore_index=True).reset_index()\ndf = df[[\"path\", \"id\", \"label\"]]\ndf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['image'] = df['path'].map(imread)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image = (df['image'][500], df['label'][500])\n_ = plt.imshow(image[0])\n_ = plt.title(image[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_images = np.stack(list(df.image), axis=0)\ninput_images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y = LabelBinarizer().fit_transform(df.label)\nX = input_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_X, test_X, train_Y, test_Y = train_test_split(X, Y, test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras import layers\nfrom keras.layers import Input, Dense, Activation, ZeroPadding2D, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout, GlobalMaxPooling2D, GlobalAveragePooling2D\nfrom keras.models import Model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def model(input_shape):\n    # Defining the input placeholder\n    X_input = Input(input_shape)\n    \n    # Padding the borders\n    X = ZeroPadding2D((3, 3))(X_input)\n    \n    # Applying the first block\n    X = Conv2D(32, (7, 7), strides= (1, 1), name='conv0')(X)\n    X = BatchNormalization(axis=3, name='bn0')(X)\n    X = Activation('relu')(X)\n    \n    # MaxPool\n    X = MaxPooling2D((2, 2), name='max_pool1')(X)\n    \n    # Applying the second block\n    X = Conv2D(64, (7, 7), strides= (1, 1), name='conv1')(X)\n    X = BatchNormalization(axis=3, name='bn1')(X)\n    X = Activation('relu')(X)\n    \n    # MaxPool\n    X = MaxPooling2D((2, 2), name='max_pool2')(X)\n      \n    # Applying the third block\n    X = Conv2D(128, (7, 7), strides= (1, 1), name='conv2')(X)\n    X = BatchNormalization(axis=3, name='bn2')(X)\n    X = Activation('relu')(X)\n    \n    # MaxPool\n    X = MaxPooling2D((2, 2), name='max_pool3')(X)  \n    \n    \n    # Flatten and FullyConnected Layer\n    X = Flatten()(X)\n    X = Dense(1, activation='sigmoid', name='fc')(X)\n    \n    model = Model(inputs=X_input, outputs=X, name='Model')\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final = model(train_X.shape[1:])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final.compile('adam', 'binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_final.fit(train_X, train_Y, epochs=10, batch_size=50)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"evals = model_final.evaluate(test_X, test_Y, batch_size=32, verbose=1)\n\nprint('Test accuracy: '+str(evals[1]*100)+'%')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data = pd.DataFrame({'path': glob(os.path.join('../input/test', '*.tif'))})\ntest_data['id'] = test_data.path.map(lambda x: x.split('/')[3].split(\".\")[0])\ntest_data['image'] = test_data['path'].map(imread)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images = np.stack(test_data.image, axis=0)\ntest_images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_labels = [model_final.predict(np.expand_dims(tensor, axis=0))[0][0] for tensor in test_images]\npredictions = np.array(predicted_labels)\ntest_data['label'] = predictions\nsubmission = test_data[[\"id\", \"label\"]]\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False, header = True)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}