{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"**Interesting kernel... lets start**"},{"metadata":{"trusted":true,"_uuid":"4d5446fe8ce8c15bd6b0ccb526795bb295d0b130"},"cell_type":"code","source":"#-------Import Dependencies-------#\nimport pandas as pd\nimport os,shutil,math\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import classification_report\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix,roc_curve,auc\n\nfrom PIL import Image\nfrom PIL import ImageDraw\nfrom glob import glob\nfrom tqdm import tqdm\nfrom skimage.io import imread\nfrom IPython.display import SVG\n\nfrom keras.utils.vis_utils import model_to_dot\nfrom keras.applications.vgg19 import VGG19,preprocess_input\nfrom keras.applications.xception import Xception\nfrom keras.applications.nasnet import NASNetMobile\nfrom keras.models import Sequential,Input,Model\nfrom keras.layers import Dense,Flatten,Dropout,Concatenate,GlobalAveragePooling2D\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.optimizers import Adam,SGD\nfrom keras.utils.vis_utils import plot_model\nfrom keras.callbacks import ModelCheckpoint,EarlyStopping,TensorBoard,CSVLogger,ReduceLROnPlateau,LearningRateScheduler","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c29c426a5266986b2df6a8bbb60018256ab6822d"},"cell_type":"code","source":"#-----In case you want to use a learning rate scheduler from keras this is a good step decay function to play around with-----#\ndef step_decay(epoch):\n    initial_lrate=0.1\n    drop=0.6\n    epochs_drop = 3.0\n    lrate= initial_lrate * math.pow(drop,math.floor((1+epoch)/epochs_drop))\n    return lrate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54cd52674429071360f38ec32251bfe1f1c4a3d6"},"cell_type":"code","source":"#----Custom function to visualize the training of the model------#\ndef show_final_history(history):\n    fig, ax = plt.subplots(1, 2, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('acc')\n    ax[1].plot(history.epoch, history.history[\"acc\"], label=\"Train acc\")\n    ax[1].plot(history.epoch, history.history[\"val_acc\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a93cab641c9e4ede8a280b248aa31dbc2e8acd1"},"cell_type":"code","source":"TRAINING_LOGS_FILE = \"training_logs.csv\"\nMODEL_SUMMARY_FILE = \"model_summary.txt\"\nMODEL_FILE = \"histopathologic_cancer_detector.h5\"\nTRAINING_PLOT_FILE = \"training.png\"\nVALIDATION_PLOT_FILE = \"validation.png\"\nROC_PLOT_FILE = \"roc.png\"\nKAGGLE_SUBMISSION_FILE = \"kaggle_submission.csv\"\nINPUT_DIR = '../input/'\nSAMPLE_COUNT = 60000\nTESTING_BATCH_SIZE = 5000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a446de42aaff740f582ccb1ac9ebca29e56c7a3"},"cell_type":"code","source":"training_dir = INPUT_DIR + 'train/'\n\ndf = pd.DataFrame({'path': glob(os.path.join(training_dir,'*.tif'))})\n\ndf['id'] = df.path.map(lambda x: x.split('/')[3].split('.')[0])\n\nlabels = pd.read_csv(INPUT_DIR + 'train_labels.csv')\n\ndf = df.merge(labels,on='id')\n\nnegative_values = df[df.label == 0].sample(SAMPLE_COUNT)\npositive_values = df[df.label == 1].sample(SAMPLE_COUNT)\n\ndf = pd.concat([negative_values,positive_values]).reset_index()\n\ndf = df[['path','id','label']]\ndf['image'] = df['path'].map(imread)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7bd2cbd566abe13abd16784508336b8651b716c"},"cell_type":"code","source":"train_path = '../training'\nval_path = '../validation'\n\nfor directory in [train_path,val_path]:\n    for sub_directory in ['0','1']:\n        path = os.path.join(directory,sub_directory)\n        os.makedirs(path,exist_ok=True)\n        \ntrain,val = train_test_split(df,train_size=0.8,stratify=df['label'])\ndf.set_index('id',inplace=True)\n\nfor images_paths in [(train,train_path),(val,val_path)]:\n    images = images_paths[0]\n    path = images_paths[1]\n    for image in images['id'].values:\n        file_name = image + '.tif'\n        label = str(df.loc[image,'label'])\n        destination = os.path.join(path,label,file_name)\n        if not os.path.exists(destination):\n            source = os.path.join(INPUT_DIR + 'train',file_name)\n            shutil.copyfile(source,destination)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b6ae8596a98a360716acbbf0f44c5d701e84707"},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3776f0b1fe89f3642166619a3cf6e1d0c16aad54"},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c6545ccd7dd99b142f98bcfd3382246b54efd16"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}