{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Check folder\nprint(os.listdir(\"../input\"))\n\n# Save train labels to dataframe\ndf = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\n\n# Remove error image\ndf = df[df['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n\n# Remove error black image\ndf = df[df['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\n\n# Show first rows\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# True positive diagnosis\ndf_true_positive = df[df[\"label\"] == 1]\nprint(\"True positive diagnosis: \" + str(len(df_true_positive)))\n\n# True negative diagnosis\ndf_true_negative = df[df[\"label\"] == 0]\nprint(\"True negative diagnosis: \" + str(len(df_true_negative)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2c80315c59bbc006cb21876ba13ada90536489f"},"cell_type":"code","source":"# Undersampling of dominant class to reach a 1/1 balance\n\nfrom sklearn.utils import shuffle\n\ndf_true_negative = shuffle(df_true_negative)\ndf_true_negative = df_true_negative[0:88800]\n\ndf_true_positive = shuffle(df_true_positive)\ndf_true_positive = df_true_positive[0:88800]\n\n# concat the dataframes\ndf = pd.concat([df_true_negative, df_true_positive], axis=0)\n\n# shuffle\ndf = shuffle(df)\ndf['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d283129394822a2f7eb1f76fa93460b78ba1c4e5"},"cell_type":"code","source":"# Split data set  to train and validation sets\nfrom sklearn.model_selection import train_test_split\n\n# Use stratify= df['label'] to get balance ratio 1/1 in train and validation sets\ndf_train, df_val = train_test_split(df, test_size=0.2, stratify= df['label'])\n\n# Check balancing\nprint(\"True positive in train data: \" +  str(len(df_train[df_train[\"label\"] == 1])))\nprint(\"True negative in train data: \" +  str(len(df_train[df_train[\"label\"] == 0])))\nprint(\"True positive in validation data: \" +  str(len(df_val[df_val[\"label\"] == 1])))\nprint(\"True negative in validation data: \" +  str(len(df_val[df_val[\"label\"] == 0])))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95b130811f3e28c35ea56f4ad185c95db1bd2581"},"cell_type":"code","source":"# Delete directory\nimport shutil\nshutil.rmtree('main', ignore_errors=True)\n\n# Create directory\nos.mkdir('main')\n\n# Create subfolder for train and val images\nos.mkdir(os.path.join('main', 'train'))\nos.mkdir(os.path.join('main', 'val'))\n\n# Create subfolders for true positive and true negative in train\nos.mkdir(os.path.join('main','train','true_positive'))\nos.mkdir(os.path.join('main','train','true_negative'))      \n         \n# Create subfolders for true positive and true negative in val\nos.mkdir(os.path.join('main','val','true_positive'))\nos.mkdir(os.path.join('main','val','true_negative'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4f40cab381e9afb2f66aac7df487c10f645e838"},"cell_type":"code","source":"# Prepare image name classes for the directory structure\n# Save all train true positive names to list and add .tif\ntrain_true_positive = df_train[df_train[\"label\"] == 1]['id'].tolist()\ntrain_true_positive = [name + \".tif\" for name in train_true_positive]\n\n# Save all train true negativeto names list and add .tif\ntrain_true_negative = df_train[df_train[\"label\"] == 0]['id'].tolist()\ntrain_true_negative = [name + \".tif\" for name in train_true_negative]\n\n# Save all val true positive \"id\" to list and add .tif\nval_true_positive = df_val[df_val[\"label\"] == 1]['id'].tolist()\nval_true_positive = [name + \".tif\" for name in val_true_positive]\n\n# Save all val true negative \"id\" to list\nval_true_negative = df_val[df_val[\"label\"] == 0]['id'].tolist()\nval_true_negative = [name + \".tif\" for name in val_true_negative]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa7de87e0b5988de60ef4b91b21c9a7859c09498"},"cell_type":"code","source":"from PIL import Image\nfrom PIL import ImageDraw\nfrom sklearn.utils import shuffle\nimport matplotlib.pyplot as plt\n\nimage_numbers = 3 \nfig = plt.figure()\nfig, ax = plt.subplots(2,image_numbers, figsize=(18,8))\n\nfor i in range(0,image_numbers):\n    \n    # Random images with true positive diagnosis\n    pos_train_names = shuffle(val_true_positive)\n    image = Image.open(os.path.join(\"../input/histopathologic-cancer-detection/train\",pos_train_names[i]))\n    print(image.size)\n    draw = ImageDraw.Draw(image)\n    draw.rectangle([(32,32),(64,64)], outline=\"yellow\")\n    ax[0,i].imshow(image)\n    ax[0,i].set_title(\"True Positive\",fontsize=14)\n    \n    # Random images with true negative diagnosis\n    neg_train_names = shuffle(train_true_negative)\n    image = Image.open(os.path.join(\"../input/histopathologic-cancer-detection/train\",neg_train_names[i]))\n    draw = ImageDraw.Draw(image)\n    draw.rectangle([(32,32),(64,64)], outline=\"yellow\")\n    ax[1,i].imshow(image)\n    ax[1,i].set_title(\"True Negative\",fontsize=14)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d838cf6b1bca8e4788856b17d1defb5be9a7d82"},"cell_type":"code","source":"# Move images to directory structure\nimport shutil\nimport os\nfrom tqdm import tqdm\n\ndef transfer(source,destination,files):\n    for image in tqdm(files):\n        # source path to image\n        src = os.path.join(source,image)\n        dst = os.path.join(destination,image)\n        # copy the image from the source to the destination\n        shutil.copyfile(src,dst)\n        \n# transfer\ntransfer('../input/histopathologic-cancer-detection/train','main/train/true_positive',train_true_positive)\ntransfer('../input/histopathologic-cancer-detection/train','main/train/true_negative',train_true_negative)\ntransfer('../input/histopathologic-cancer-detection/train','main/val/true_positive',val_true_positive)\ntransfer('../input/histopathologic-cancer-detection/train','main/val/true_negative',val_true_negative)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7595af5b78ef78e0a9170a7c72400c94853c018b"},"cell_type":"code","source":"# Import VGG16 model, with weights pre-trained on ImageNet.\nfrom keras.applications.vgg16 import VGG16, preprocess_input\n\n# VGG model without the last classifier layers (include_top = False)\nvgg16_model = VGG16(include_top = False,\n                    input_shape = (96,96,3),\n                    #weights='../input/VGG16weights/vgg16_weights_tf_dim_ordering_tf_kernels_notop.h5')\n                    weights = '../input/vgg16weights/vgg16_weights_tf_dim_ordering_tf_kernels_notop.h5')\n    \n# Freeze the layers \nfor layer in vgg16_model.layers[:-12]:\n    layer.trainable = False\n    \n# Check the trainable status of the individual layers\nfor layer in vgg16_model.layers:\n    print(layer, layer.trainable)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e5914fe39d1e12264afa8a952dae43653b34afa"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense,Flatten,Dropout\n\nmodel = Sequential()\nmodel.add(vgg16_model)\nmodel.add(Flatten())\nmodel.add(Dense(1024, activation=\"relu\"))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(512, activation=\"relu\"))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(2, activation=\"softmax\"))\n# Load previous weights to save the time\n#model.load_weights('../input/pretrained3/model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10976da747dbcd9a2747572e894baaf0ef4ea211"},"cell_type":"code","source":"# Generate batches of tensor image data with real-time data augmentation. \nimport numpy as np\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 32\nval_batch_size = 32\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n\nprint(train_steps)\nprint(val_steps)\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Augmentation \ntrain_datagen = ImageDataGenerator(\n                rescale=1./255,\n                vertical_flip=True,\n                horizontal_flip=True,\n                rotation_range=90,\n                shear_range=0.05)\n\n# Augmentation \n# train_datagen = ImageDataGenerator(\n#                 rescale=1./255,\n#                 vertical_flip=True, # set True or False\n#                 horizontal_flip=True, # set True or False\n#                 rotation_range=180, # 0-180 range, degrees of rotatation\n#                 shear_range=0.01,# 0-1 range for shearing\n#                 width_shift_range=0.2, # 0-1 range, horizontal translation\n#                 height_shift_range=0.2) # 0-1 range vertical translation\n\n#train_datagen = ImageDataGenerator(rescale=1./255)\n\n# Augmentation configuration for validatiopn and testing: only rescaling!\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\n# Generator that will read pictures found in subfolers of 'main/train', and indefinitely generate batches of augmented image data\ntrain_generator = train_datagen.flow_from_directory('main/train',\n                                            target_size=(96,96),\n                                            batch_size=train_batch_size,\n                                            class_mode='categorical')\n\nval_generator = test_datagen.flow_from_directory('main/val',\n                                            target_size=(96,96),\n                                            batch_size=val_batch_size,\n                                            class_mode='categorical')\n\n# !!! batch_size=1 & shuffle=False !!!!\ntest_generator = test_datagen.flow_from_directory('main/val',\n                                            target_size=(96,96),\n                                            batch_size=1,\n                                            class_mode='categorical',\n                                            shuffle=False)\n\n# Get the labels that are associated with each index\nprint(val_generator.class_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c03c409be15b432b50bd761df7fde9af5fd233b8"},"cell_type":"code","source":"from keras.optimizers import Adam, SGD\nfrom keras import optimizers\n\n#model.compile(Adam(lr=0.00001), loss='binary_crossentropy', metrics=['acc'])\nmodel.compile(loss='binary_crossentropy',optimizer=optimizers.SGD(lr=0.00001, momentum=0.95),metrics=['accuracy'])\n\n# Due to Disk limits the saving of best weights isn't possible during the training process                                            \nhistory = model.fit_generator(\n                    train_generator, \n                    steps_per_epoch  = train_steps, \n                    validation_data  = val_generator,\n                    validation_steps = val_steps,\n                    epochs           = 20, \n                    verbose          = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8bee397c1b5b8dcc16dd746a7379268f4f09dd26"},"cell_type":"code","source":"# Plot validation and accuracies over epochs\nimport matplotlib.pyplot as plt\n\ntrain_acc = history.history['acc']\nval_acc = history.history['val_acc']\n\nepochs = range(len(train_acc))\n\nplt.plot(epochs,train_acc,'b',label='Training accuracy')\nplt.plot(epochs,val_acc,'r',label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85e3bdceaa5c6af52d4e898d1fca0e20128ad41a"},"cell_type":"code","source":"# Due to the disk limits I couldn't save the best model during the training process\nprint(\"Validation Accuracy: \" + str(history.history['val_acc'][-1:]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85fa69fca8abd656db0250c78f68ae9a5d05a9b3"},"cell_type":"code","source":"# Prediction on validation data sets\nval_predict = model.predict_generator(test_generator, steps=len(df_val), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b383e2509f37bcdb9b5f8dd3ec1ee028e48ded0"},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nprint(\"Confusion matrix is: \")\nprint(confusion_matrix(test_generator.classes, val_predict.argmax(axis=1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4aef05b7cbfc32850679c32eba196be692c372fa"},"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\nfpr, tpr, thresholds = roc_curve(test_generator.classes, val_predict.argmax(axis=1))   \n# Compute ROC area\nprint(\"ROC area is: \" + str(auc(fpr, tpr)))\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkred', label='ROC curve (area = %0.2f)' % auc(fpr, tpr))\nplt.plot([0, 1], [0, 1], color='darkblue', linestyle='--')\nplt.xlim([-0.01, 1.0])\nplt.ylim([0.0, 1.01])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ded48bc0d5c52884bb05dbebc896da09e13471a1"},"cell_type":"code","source":"from tqdm import tqdm\n\n# Delete directory structure\nshutil.rmtree('main')\n\n# Create test folder\nos.mkdir('test')\n    \n# Create test_images folder inside test folder\nos.mkdir(os.path.join('test','test_images'))\n\n# Save test image identification names\ntest_images = os.listdir('../input/histopathologic-cancer-detection/test')\n\n# Move images to test folder\nfor test_image in tqdm(test_images):   \n    # source \n    src = os.path.join('../input/histopathologic-cancer-detection/test',test_image)\n    # destination \n    dst = os.path.join('test/test_images',test_image)\n    # copy the image\n    shutil.copyfile(src, dst)\n    \n# !!! batch_size=1 & shuffle=False !!!!\ntest_generator = test_datagen.flow_from_directory('test',\n                                            target_size=(96,96),\n                                            batch_size=1,\n                                            class_mode='categorical',\n                                            shuffle=False)\n\n# model.load_weights('best_model.h5')\n# Predict \ntest_predict = model.predict_generator(test_generator, steps=57458, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"546d0aba1eaccec01fb2f797b4e0d5f4d07aec31"},"cell_type":"code","source":"# Extract test names from test_generator\ntest_names = [name.split(\"/\")[1].split(\".\")[0] for name in test_generator.filenames]\n\n# Create a dataframe\ndf_pred_test = pd.DataFrame(test_names, columns=[\"id\"])\n\n# !!! label == probability of true positive (has cancer) !!!\ndf_pred_test[\"label\"] = pd.DataFrame(test_predict[:,1])\n\ndf_pred_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd70b00ae37462afefc7d8fbfd9d6ec2f03bda20"},"cell_type":"code","source":"# Save sample submissions as dataframe\ndf_samples = pd.read_csv('../input/histopathologic-cancer-detection/sample_submission.csv')\n\n# Merge df_samples and df_pred_test using unique key combination through \"id\"\nsubmission = pd.merge(df_samples.drop(\"label\",axis=1), df_pred_test, on = 'id')\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74c99ed0324192575264fe77b82898429e3240cd"},"cell_type":"code","source":"shutil.rmtree('test')\n\n# Save submission file\nsubmission.to_csv(\"submission.csv\",index=False) \n\n# Due to the disk limits I can only save the last model\nmodel.save('model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ca6a4ed3edd8f824c74aa3c11de0dac4ab74387"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}