{"cells":[{"metadata":{"trusted":true,"_uuid":"5e259e7949b31aa129c9bd41061de96810e63b41"},"cell_type":"code","source":"import h5py\nimport pandas as pd\nimport numpy as np\nimport os\nimport math\nfrom glob import glob\nfrom sklearn.utils import shuffle\nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom keras.initializers import glorot_uniform\nfrom keras.models import Model, load_model, Sequential\nfrom keras import optimizers\nfrom keras import regularizers\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.applications.inception_v3 import InceptionV3\nfrom keras.layers import Input, Add, Dense, Activation,GlobalAveragePooling2D, ZeroPadding2D, BatchNormalization, Flatten, Conv2D, AveragePooling2D, MaxPooling2D, GlobalMaxPooling2D, Dropout\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import TensorBoard, ReduceLROnPlateau\nfrom keras.utils.vis_utils import plot_model\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom IPython.display import clear_output\n","execution_count":1,"outputs":[{"output_type":"stream","text":"Using TensorFlow backend.\n","name":"stderr"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')\n#df[df['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n#df[df['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\ndf_0 = df[df['label'] == 0].sample(89000, random_state = 101)\ndf_1 = df[df['label'] == 1].sample(89000, random_state = 101)\ndf = pd.concat([df_0, df_1], axis=0).reset_index(drop=True)\ndf = shuffle(df)\ndf['id'] =df.id.map(lambda x:x +'.tif')\ndf['label'] = df['label'].astype(str)\ndf_sample = df.sample(n=10000, random_state=2018)","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5fb763a63e26d0e00bd75f5c8344ad9e341c0342"},"cell_type":"code","source":"test_df = pd.DataFrame({'filename':os.listdir('../input/histopathologic-cancer-detection/test')})\n\n","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59e41b940bd7ea0ceab87f67d3885ee2771857e8"},"cell_type":"code","source":"clear_output()\nimage_gen = ImageDataGenerator( samplewise_center = True, brightness_range = (0.1,0.2),zca_whitening = True,\n                validation_split=0.2, vertical_flip = True, horizontal_flip = True)\nimage_gen_test = ImageDataGenerator( vertical_flip = True, horizontal_flip = True, brightness_range = (0.1,0.2),\n                  samplewise_center = True, zca_whitening = True)","execution_count":4,"outputs":[{"output_type":"stream","text":"/opt/conda/lib/python3.6/site-packages/keras_preprocessing/image/image_data_generator.py:334: UserWarning: This ImageDataGenerator specifies `zca_whitening`, which overrides setting of `featurewise_center`.\n  warnings.warn('This ImageDataGenerator specifies '\n","name":"stderr"}]},{"metadata":{"trusted":true,"_uuid":"a39c05cd8ebdbc0a5ed7d8e07ea66c032934d39d","scrolled":true},"cell_type":"code","source":"\n\nclear_output()\nimg_iter = image_gen.flow_from_dataframe(\n    df_sample,\n    shuffle=True,\n    directory= '../input/histopathologic-cancer-detection/train',\n    x_col='id',\n    y_col='label',\n    class_mode='binary',\n    color_mode = 'rgb',\n    target_size=(96, 96),\n    batch_size=256,\n    subset='training'\n)\n\nimg_iter_val = image_gen.flow_from_dataframe(\n    df_sample,\n    shuffle=False,\n    directory= '../input/histopathologic-cancer-detection/train',\n    x_col='id',\n    y_col='label',\n    class_mode='binary',\n    color_mode = 'rgb',\n    target_size=(96, 96),\n    batch_size=256,\n    subset='validation'\n)\n\n\n#test_generator = image_gen_test.flow_from_dataframe(\n        #dataframe = test_df,\n        #directory = '../input/histopathologic-cancer-detection/test',\n        #target_size=(96, 96),\n        #color_mode = 'rgb',\n        #shuffle = False,\n        #class_mode=None,\n        #batch_size=256)","execution_count":5,"outputs":[{"output_type":"stream","text":"Found 8000 images belonging to 2 classes.\nFound 2000 images belonging to 2 classes.\nFound 57458 images.\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nbase_model = InceptionV3(weights = None, input_shape = (96,96,3), include_top=False)\n# add a global spatial average pooling layer\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n# let's add a fully-connected layer\nx = Dense(1024, activation='relu')(x)\n# and a logistic layer -- let's say we have 200 classes\nx = Dense(200, activation='relu')(x)\npredictions = Dense(1, activation = 'sigmoid')(x)\n\n# this is the model we will train\nmodel = Model(inputs=base_model.input, outputs=predictions)","execution_count":7,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"81d96f67c2624e488d2b09306d68f26f22c233d1"},"cell_type":"code","source":"clear_output()\n#model = my_model\nadam = optimizers.Adam(lr=0.00007)\nmodel.compile(optimizer=adam, loss='binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83247ad45c58640c1ae84660a20f3a9c7db02971"},"cell_type":"code","source":"\nclear_output()\nSTEP_SIZE_TRAIN=np.ceil(img_iter.n//img_iter.batch_size)+1\nSTEP_SIZE_VALID=np.ceil(img_iter_val.n//img_iter_val.batch_size)+1\n#STEP_SIZE_TEST=np.ceil(test_generator.n//test_generator.batch_size)+1\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(img_iter, steps_per_epoch=STEP_SIZE_TRAIN,\n                    epochs= 13, validation_data = img_iter_val, validation_steps = STEP_SIZE_VALID)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_final_history(history):\n    fig, ax = plt.subplots(1, 2, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('acc')\n    ax[1].plot(history.epoch, history.history[\"acc\"], label=\"Train acc\")\n    ax[1].plot(history.epoch, history.history[\"val_acc\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_final_history(history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_json = model.to_json()\nmodel.save(\"model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0533e5f1d8753c6faa3b5910d6538a82d3a8519b"},"cell_type":"markdown","source":"\ntest_generator.reset()\npred=model.predict_generator(test_generator, steps= STEP_SIZE_TEST, verbose=0)\n"},{"metadata":{"trusted":true},"cell_type":"markdown","source":"print(len(os.listdir('../input/histopathologic-cancer-detection/test')))\nprint(len(pred))"},{"metadata":{"trusted":true,"_uuid":"98ff00fcf5f8ac020e626e442d4a4a75250b22b1"},"cell_type":"markdown","source":"pred_rounded = [int(round(pred[i][0])) for i in range(0, pred.shape[0])]\n\nfilenames=test_generator.filenames\nfilenames = [f.split(sep='.')[0] for f in filenames]\n\n\n\nresults=pd.DataFrame({\"id\":filenames,\n                      \"label\":pred_rounded})\nresults.to_csv(\"results.csv\",index=False)"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}