{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nfrom glob import glob \nfrom skimage.io import imread \nimport os\nimport shutil\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dropout, Flatten, Dense, GlobalAveragePooling2D, Input, Concatenate, GlobalMaxPooling2D\nfrom keras.models import Model\nfrom keras.callbacks import CSVLogger, ReduceLROnPlateau, ModelCheckpoint\nfrom keras.optimizers import Adam\n!pip install livelossplot\nfrom livelossplot import PlotLossesKeras\nfrom keras.applications.vgg19 import VGG19\nfrom keras.applications.vgg16 import VGG16","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAINING_LOGS_FILE = \"training_logs.csv\"\nMODEL_FILE = \"histopathologic_cancer_detector.h5\"\nROC_PLOT_FILE = \"roc.png\"\nKAGGLE_SUBMISSION_FILE = \"kaggle_submission.csv\"\nINPUT_DIR = '../input/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SAMPLE_COUNT = 50000\nTRAINING_RATIO = 0.9\nIMAGE_SIZE = 96\nEPOCHS = 12\nBATCH_SIZE = 128\nVERBOSITY = 1\nTESTING_BATCH_SIZE = 1000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Data setup\ntraining_dir = INPUT_DIR + 'train/'\ndata_frame = pd.DataFrame({'path': glob(os.path.join(training_dir,'*.tif'))})\ndata_frame['id'] = data_frame.path.map(lambda x: x.split('/')[3].split('.')[0]) \nlabels = pd.read_csv(INPUT_DIR + 'train_labels.csv')\ndata_frame = data_frame.merge(labels, on = 'id')\nnegatives = data_frame[data_frame.label == 0].sample(SAMPLE_COUNT)\npositives = data_frame[data_frame.label == 1].sample(SAMPLE_COUNT)\ndata_frame = pd.concat([negatives, positives]).reset_index()\ndata_frame = data_frame[['path', 'id', 'label']]\ndata_frame['image'] = data_frame['path'].map(imread)\n\ntraining_path = '../training1'\nvalidation_path = '../validation1'\n\nfor folder in [training_path, validation_path]:\n    for subfolder in ['0', '1']:\n        path = os.path.join(folder, subfolder)\n        os.makedirs(path, exist_ok=True)\n\ntraining, validation = train_test_split(data_frame, train_size=TRAINING_RATIO, stratify=data_frame['label'])\n\ndata_frame.set_index('id', inplace=True)\n\nfor images_and_path in [(training, training_path), (validation, validation_path)]:\n    images = images_and_path[0]\n    path = images_and_path[1]\n    for image in images['id'].values:\n        file_name = image + '.tif'\n        label = str(data_frame.loc[image,'label'])\n        destination = os.path.join(path, label, file_name)\n        if not os.path.exists(destination):\n            source = os.path.join(INPUT_DIR + 'train', file_name)\n            shutil.copyfile(source, destination)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Augmentation of data\ntraining_data_generator = ImageDataGenerator(rescale=1./255,\n                                             \n                                             horizontal_flip=True,\n                                             vertical_flip=True,\n                                             rotation_range=180,\n                                             zoom_range=[1, 1.5], \n                                             width_shift_range=0.3,\n                                             height_shift_range=0.3,\n                                             shear_range=0.3,\nchannel_shift_range=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_generator = training_data_generator.flow_from_directory(training_path,\n                                                                 target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                                                 batch_size=BATCH_SIZE,\n                                                                 class_mode='binary')\nvalidation_generator = ImageDataGenerator(rescale=1./255).flow_from_directory(validation_path,\n                                                                              target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                                                              batch_size=BATCH_SIZE,\n                                                                              class_mode='binary')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model\ninput_shape = (96, 96, 3)\ninputs = Input(input_shape)\nv16 = VGG16(include_top=False, input_shape=input_shape)\nv19 = VGG19(include_top=False, input_shape=input_shape) \noutputs = Concatenate(axis=-1)([GlobalAveragePooling2D()(v16(inputs)),GlobalAveragePooling2D()(v19(inputs))])\n#outputs = GlobalAveragePooling2D()(dense(inputs))\noutputs = Dropout(0.5)(outputs)\noutputs = Dense(1, activation='sigmoid')(outputs)\n\nmodel = Model(inputs, outputs)\nmodel.compile(optimizer=Adam(lr=0.0001, decay=0.00001),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\nmodel.summary()        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#  Training\nhistory = model.fit_generator(training_generator,\n                              steps_per_epoch=len(training_generator), \n                              validation_data=validation_generator,\n                              validation_steps=len(validation_generator),\n                              epochs=EPOCHS,\n                              verbose=VERBOSITY,shuffle=True,\n                              callbacks=[PlotLossesKeras(),\n                                         ModelCheckpoint(MODEL_FILE,\n                                                         monitor='val_acc',\n                                                         verbose=VERBOSITY,\n                                                         save_best_only=True,\n                                                         mode='max'),\n                                         CSVLogger(TRAINING_LOGS_FILE,\n                                                   append=False,\n                                                   separator=';')])\nmodel.load_weights(MODEL_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Test on validation data\nroc_validation_generator = ImageDataGenerator(rescale=1./255,shear_range=0.3,zoom_range=[1,1.5],\n        horizontal_flip=True,\n        rotation_range=10., \n        width_shift_range = 0.3, \n        height_shift_range = 0.3).flow_from_directory(validation_path,target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                                                                  batch_size=BATCH_SIZE,\n                                                                                  class_mode='binary',\n                                                                                  shuffle=False)\npredictions = model.predict_generator(roc_validation_generator, steps=len(roc_validation_generator), verbose=VERBOSITY)\nfalse_positive_rate, true_positive_rate, threshold = roc_curve(roc_validation_generator.classes, predictions)\narea_under_curve = auc(false_positive_rate, true_positive_rate)\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(false_positive_rate, true_positive_rate, label='AUC = {:.3f}'.format(area_under_curve))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.savefig(ROC_PLOT_FILE, bbox_inches='tight')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Kaggle testing\ntta_steps = 5\nfrom tqdm import tqdm\ntest_datagen = ImageDataGenerator(\n        shear_range=0.1,\n        zoom_range=[1,1.5],\n        horizontal_flip=True,\n        rotation_range=10.,\n        fill_mode='reflect', \n        width_shift_range = 0.1, \n        height_shift_range = 0.1)\ntesting_files = glob(os.path.join(INPUT_DIR+'test/','*.tif'))\nsubmission = pd.DataFrame()\nfor index in range(0, len(testing_files), TESTING_BATCH_SIZE):\n    predicted_labels=[]\n    data_frame = pd.DataFrame({'path': testing_files[index:index+TESTING_BATCH_SIZE]})\n    data_frame['id'] = data_frame.path.map(lambda x: x.split('/')[3].split(\".\")[0])\n    data_frame['image'] = data_frame['path'].map(imread)\n    if(index==57000):\n        TESTING_BATCH_SIZE=548;\n    images = np.stack(data_frame.image, axis=0)\n    for i in tqdm(range(tta_steps)):\n        preds = model.predict_generator(test_datagen.flow(images/255, batch_size=TESTING_BATCH_SIZE, shuffle=False), steps = len(images)/TESTING_BATCH_SIZE)\n        predicted_labels.append(preds)\n    predictions = np.max(predicted_labels,axis=0)\n    data_frame['label'] = predictions\n    submission = pd.concat([submission, data_frame[[\"id\", \"label\"]]])\nsubmission.to_csv(KAGGLE_SUBMISSION_FILE, index=False, header=True)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}