{"cells":[{"metadata":{"_uuid":"60e427537d2fdc1e500088a1a8b7ad09eaf3311c"},"cell_type":"markdown","source":"# **Note**\nIn the notebook, the model is first trained with a validation subset so the best model can be saved. Once the model is saved and the notebook is restarted, the model will be retrained with a small learning rate on all available data."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom keras.preprocessing import image\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom keras.models import Model, load_model\nfrom keras.optimizers import Adam\nfrom keras.applications.resnet50 import ResNet50\nfrom keras import layers as KL\nfrom keras.callbacks import ReduceLROnPlateau, ModelCheckpoint","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"'''TRAIN_PATH = '/kaggle/input/histopathologic-cancer-detection/train/'\nTRAIN_LABELS = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\nSIZE_IMG = 96\nEPOCHS = 10\n\nmodel_path = '../input/resnet-cancer-detection/cancer_detection_resnet.h5'\nsaved_model = os.path.isfile(model_path)'''\n\nTRAIN_PATH = '../input/cartraintest2/train/train/'\nTRAIN_LABELS = '../input/imagepropertiesdf6/train6.xls'\nSIZE_IMG = 96\nEPOCHS = 10\n\nmodel_path = '../input/resnet-cancer-detection/cancer_detection_resnet.h5'\nsaved_model = os.path.isfile(model_path)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"476f11b88e57a0f13ea2f078a70c4a0839daa823"},"cell_type":"markdown","source":"# **Data processing**"},{"metadata":{"trusted":true,"_uuid":"46162f40409aa15d8a6d7f5e5fb84fdc35762150"},"cell_type":"code","source":"df = pd.read_csv(TRAIN_LABELS)\n\n#remove unwanted data detected by other kaggle users\n#df = df[df['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n#df = df[df['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\n\n#print(df['label'].value_counts(), \nprint(df['model'].value_counts(), \n      '\\n\\n', df.describe(), \n      '\\n\\n', df.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a427a43773819d6901b053c487469ef680d12762"},"cell_type":"code","source":"def display_random_data(dataframe, path, rows):\n\n    imgs = dataframe.sample(rows *2)\n    fig, axarr = plt.subplots(2, rows, figsize=(rows*10, rows*4))\n\n    for i in range(1,rows*2+1):\n        img_path = path + imgs.iloc[i-1]['newFileName']# + '.tif' #'id'\n        img = image.load_img(img_path, target_size=(96,96)) #96,96\n        img = image.img_to_array(img)/255\n        axarr[i//(rows+1),i%rows].imshow(img)\n        axarr[i//(rows+1),i%rows].set_title(imgs.iloc[i-1]['model'], fontsize=35) #'label'\n        axarr[i//(rows+1),i%rows].axis('off')\n        \ndisplay_random_data(df,TRAIN_PATH, 5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1015414495b3932afb6524f3f93f4acd6da7d516"},"cell_type":"markdown","source":"# **Init Keras data generator**"},{"metadata":{"trusted":true,"_uuid":"63d6e94e60cebb877a0f392ec127c309ed28b0b8"},"cell_type":"code","source":"#add .tif to ids in the dataframe to use flow_from_dataframe\n#df[\"id\"]=df[\"id\"].apply(lambda x : x +\".tif\")\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_PATH #= '../input/cartraintest2/train/train/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e7d918f45497ad5516cfe35f276614d1d84425f"},"cell_type":"code","source":"if saved_model:\n    val = 0\nelse:\n    val = 0.15\n    \ndatagen= ImageDataGenerator(\n            rescale=1./255,\n            samplewise_std_normalization= True,\n            horizontal_flip=True,\n            vertical_flip=True,\n            rotation_range=90,\n            zoom_range=0.2, \n            width_shift_range=0.1,\n            height_shift_range=0.1,\n            shear_range=0.05,\n            channel_shift_range=0.1,\n            validation_split=val)\n\ntrain_generator=datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=TRAIN_PATH,\n    x_col=\"newFileName\",\n    y_col=\"model\",\n    subset=\"training\",\n    batch_size=64,\n    shuffle=True,\n    class_mode=\"categorical\",\n    target_size=(96,96))\n\nprint(TRAIN_PATH+df.newFileName)\n\nvalid_generator=datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=TRAIN_PATH,\n    x_col=\"newFileName\",\n    y_col=\"model\",\n    subset=\"validation\",\n    batch_size=64,\n    shuffle=True,\n    class_mode=\"categorical\",\n    target_size=(96,96))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"56ff4b7f7be00f145721114cd4052ab12c1b68b3"},"cell_type":"markdown","source":"# **Build model**"},{"metadata":{"trusted":true,"_uuid":"ca31c48aa3ad33c6b8f5f774724dd514136e54ba"},"cell_type":"code","source":"def build_model():\n    input_shape = (SIZE_IMG, SIZE_IMG, 3)\n    inputs = KL.Input(input_shape)\n    resnet = ResNet50(include_top=False, input_shape=input_shape) \n    x  = KL.GlobalAveragePooling2D()(resnet(inputs))\n    x = KL.Dropout(0.5)(x)\n    outputs = KL.Dense(1, activation='sigmoid')(x)\n\n    return Model(inputs, outputs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"918ca5f8b774881cc11689768c578e884e2f8eba"},"cell_type":"code","source":"def first_training():\n    '''\n    train the model and save it if the val_acc test is better than the precedent epoch\n    '''\n    model = build_model()\n    \n    model.compile(optimizer=Adam(lr=0.0001, decay=0.00001),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n\n    reduce_lr = ReduceLROnPlateau(monitor='val_acc', factor=0.5, patience=2, \n                                       verbose=1, mode='max', min_lr=0.000001)\n    \n    checkpoint = ModelCheckpoint(\"resnet_cancer_detection.h5\", monitor='val_acc', verbose=1, \n                              save_best_only=True, mode='max')\n\n    history = model.fit_generator(train_generator,\n                              steps_per_epoch=train_generator.n//train_generator.batch_size, \n                              validation_data=valid_generator,\n                              validation_steps=valid_generator.n//valid_generator.batch_size,\n                              epochs=EPOCHS,\n                              callbacks=[checkpoint,reduce_lr])\n    \n    return history, model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"684c20a56c666d44d6e784f19345d19c5408ea0a"},"cell_type":"code","source":"def second_training():\n    '''\n    Tune the model using all available data and a small learning rate\n    '''\n    model = load_model(model_path)\n    \n    model.compile(optimizer=Adam(lr=0.000001, decay=0.00001),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    \n    history = model.fit_generator(train_generator,\n                              steps_per_epoch=train_generator.n//train_generator.batch_size, \n                              epochs=10)\n    \n    return history, model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9446aa5eebd6b57e067526f47e864b6545a0c5fc"},"cell_type":"code","source":"if saved_model:\n    history, model = second_training()\nelse:\n    history, model = first_training()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f042f2acc3b72bef0720f22942a834f9570e4e4b"},"cell_type":"code","source":"def analyse_results(epochs):\n    metrics = ['loss', \"acc\", 'val_loss','val_acc']\n        \n    plt.style.use(\"ggplot\")\n    (fig, ax) = plt.subplots(1, 4, figsize=(30, 5))\n    fig.subplots_adjust(hspace=0.1, wspace=0.3)\n\n    for (i, l) in enumerate(metrics):\n        title = \"Loss for {}\".format(l) if l != \"loss\" else \"Total loss\"\n        ax[i].set_title(title)\n        ax[i].set_xlabel(\"Epoch #\")\n        ax[i].set_ylabel(l.split('_')[-1])\n        ax[i].plot(np.arange(0, epochs), history.history[l], label=l)\n        ax[i].legend() \n\nif EPOCHS > 1 and saved_model == False:        \n    analyse_results(EPOCHS)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"612d836545f5f52b45576739fd202325464372b8"},"cell_type":"markdown","source":"# **Predictions**"},{"metadata":{"trusted":true,"_uuid":"9f02d4c77345f111842f95d05ae39b19b3e7ffb5"},"cell_type":"code","source":"test_path = '/kaggle/input/histopathologic-cancer-detection/test/'\ndf_test = pd.read_csv('../input/histopathologic-cancer-detection/sample_submission.csv')\ndf_test[\"id\"]=df_test[\"id\"].apply(lambda x : x +\".tif\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2d08b3a6c170339a91e0b93a773ffee82ee2856f"},"cell_type":"markdown","source":"Test generator doesn't need to be shuffled and no class_mode are passed as an argument."},{"metadata":{"trusted":true,"_uuid":"10f8a839e3ac98e3c2dfbb811f0e6e46cca87625"},"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255,\n                                 samplewise_std_normalization= True)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=df_test,\n    directory=test_path,\n    x_col=\"id\",\n    y_col=None,\n    target_size=(96, 96),\n    color_mode=\"rgb\",\n    batch_size=64,\n    class_mode=None,\n    shuffle=False,\n)  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1599d89fbf0b77c2bc94511dc195f6f47c8fcb7e"},"cell_type":"code","source":"test_generator.reset()\npred=model.predict_generator(test_generator,verbose=1).ravel()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"90770b2d3d884566e3707fbd29db1631c39a3b74"},"cell_type":"markdown","source":"# **CSV submission**\nPredictions of the test generator are not in the right order so it needs to be rearranged it in the label list before to be passed it to the submission data frame."},{"metadata":{"trusted":true,"_uuid":"28cd7b4ebe731e8bd5bb834a656a51cd0d2edb69"},"cell_type":"code","source":"results = dict(zip(test_generator.filenames, pred))\n\nlabel = []\nfor i in range(len(df_test[\"id\"])):\n    label.append(results[df_test[\"id\"][i]])\n    \ndf_test[\"id\"]=df_test[\"id\"].apply(lambda x : x[:-4])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6125561bd2cd6b7fe23bd362c276ab35529bb97"},"cell_type":"code","source":"submission=pd.DataFrame({\"id\":df_test[\"id\"],\n                      \"label\":label})\nsubmission.to_csv(\"submission.csv\",index=False)\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}