{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport keras\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.vgg19 import VGG19\nfrom keras.models import Model, Sequential\nfrom keras.layers import GlobalAveragePooling2D, Dense, Dropout, Flatten, BatchNormalization\nfrom sklearn.model_selection import train_test_split\nimport os\nfrom tqdm import tqdm\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7e4e0981ead6e97a08f77e75e856982e53129b6"},"cell_type":"code","source":"cancer_labs = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\ndef append_ext(fn):\n    return fn+\".tif\"\n\ncancer_labs[\"id\"]=cancer_labs[\"id\"].apply(append_ext)\n\nprint(\"Cancer image data set count:\", cancer_labs.shape[0])\n\ncancer_labs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c63c900a54276ade0fd0a9ff8b47764371db3d2"},"cell_type":"code","source":"class_count = cancer_labs[\"label\"].value_counts()\n\nprint(\"Positive cancer scans:\", class_count[1])\nprint(\"Positive cancer scans percent:\", round(class_count[1] / cancer_labs.shape[0], 2) * 100)\nprint(\"Negative cancer scans:\", class_count[0])\nprint(\"Negative cancer scans percent:\", round(class_count[0] / cancer_labs.shape[0], 2) * 100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fd07c835a3ffd90c5f919ce842228ecd8496ce9"},"cell_type":"code","source":"train, test = train_test_split(cancer_labs, test_size=0.2, random_state=1017)\n\nprint(\"Cancer image training set rows:\", train.shape[0])\nprint(\"Cancer image test set rows:\", test.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"batch_size = 128\n\nimage_size = (96, 96)\n\ntrain_datagen = ImageDataGenerator(\n        rescale=1./255,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True\n)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train, \n    directory='../input/histopathologic-cancer-detection/train/', \n    x_col='id', \n    y_col='label',\n    target_size=image_size,\n    batch_size=batch_size,\n    class_mode=\"binary\",\n    has_ext=False\n)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test, \n    directory='../input/histopathologic-cancer-detection/train/', \n    x_col='id', \n    y_col='label',\n    target_size=image_size,\n    batch_size=batch_size,\n    class_mode=\"binary\",\n    has_ext=False\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9bf1733b916392c6419264826ec4f7c7f78eb2a3"},"cell_type":"code","source":"weights_path = '../input/vgg19/vgg19_weights_tf_dim_ordering_tf_kernels_notop.h5'\nbase_model  = VGG19(include_top=False, weights= weights_path,input_shape=(96, 96, 3))\n\nx = base_model.output\nx = Flatten()(x)\nx = Dense(256, activation='relu', kernel_initializer='glorot_uniform', use_bias=False)(x)\nx = BatchNormalization()(x)\nx = Dropout(.5)(x)\nx = Dense(256, activation='relu', use_bias=False)(x)\nx = Dropout(.5)(x)\npredictions = Dense(1, activation = 'sigmoid')(x)\n\nmodel = Model(inputs = base_model.input, outputs = predictions)\nmodel.compile(loss=\"binary_crossentropy\",\n              optimizer=\"rmsprop\",\n              metrics=[\"binary_accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68b9765d692c60408829a340a4c6f384b641feb4","scrolled":false},"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ModelCheckpoint\nearlyStopping = EarlyStopping(monitor='val_loss', patience=10, verbose=0, mode='min')\nmcp_save = ModelCheckpoint('.mdl_wts.hdf5', save_best_only=True, monitor='val_loss', mode='min')\n\nhistory  = model.fit_generator(generator=train_generator, \n                                         epochs=25, \n                                         steps_per_epoch=512, \n                                         validation_data=test_generator, \n                                         validation_steps=128,\n                                          callbacks = [earlyStopping,mcp_save])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a18bab68b9843377169be8b1ef7f515cb3f840a8"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fbbb1b0d6eff0d04a8c37ea097d290fc7995d29"},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_history(history):\n    acc = history.history[\"binary_accuracy\"]\n    val_acc = history.history[\"val_binary_accuracy\"]\n    loss = history.history[\"loss\"]\n    val_loss = history.history[\"val_loss\"]\n    x = range(1, len(acc) + 1)\n    \n    plt.figure(figsize=(14, 7))\n    plt.subplot(1, 2, 1)\n    plt.plot(x, acc, \"b\", label=\"Training acc\")\n    plt.plot(x, val_acc, \"r\", label=\"Validation acc\")\n    plt.xticks(x, x)\n    plt.title(\"Training and validation accuracy\")\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(x, loss, \"b\", label=\"Training loss\")\n    plt.plot(x, val_loss, \"r\", label=\"Validation loss\")\n    plt.xticks(x, x)\n    plt.title(\"Training and validation loss\")\n    plt.legend()\n    \nplot_history(history=history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36b3095b734d5a4616de064b57ef4ba69f949898"},"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, roc_auc_score, f1_score\nimport matplotlib.pyplot as plt\n\nmodel_probs = model.predict_generator(test_generator, steps=len(test_generator.filenames) / 128, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88f8a815680ba0ed78be9826c85462561a4a85d1"},"cell_type":"code","source":"roc_auc_score(test_generator.classes, model_probs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c736b694bbd02090581caf2d4c38ed42c9876dc8"},"cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(test_generator.classes, model_probs)\ni = np.arange(len(tpr)) # index for df\nroc = pd.DataFrame({'fpr' : pd.Series(fpr, index=i),'tpr' : pd.Series(tpr, index = i), '1-fpr' : pd.Series(1-fpr, index = i), 'tf' : pd.Series(tpr - (1-fpr), index = i), 'thresholds' : pd.Series(thresholds, index = i)})\nroc.loc[(roc.tf-0).abs().argsort()[:1]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0335b41d3339c5b82c3b7ca44b98d6fc2f924ff"},"cell_type":"code","source":"f= roc.loc[(roc.tf-0).abs().argsort()[:1]].thresholds\ny_pred1 = np.where(model_probs > f.values[0], 1, 0)\nprint(\"F1 score is equivalent to {}\".format(f1_score(test_generator.classes,y_pred1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec2aba070390ac90a1121853bed3fa22a6446cf5"},"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(test_generator.classes,y_pred1))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}