{"cells":[{"metadata":{},"cell_type":"markdown","source":"**In this model, we use a simple Keras CNN to train from scratch**\n* We use the Keras ImageDataGenerator library for augmentation and pre-processing\n* We implement tensorboard and Checkpoint callbacks for accuracy monitoring"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Specifying Train and Test directories"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_dir = '../input/train_images'\ntest_dir = '../input/test_images'\ntrain_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(test_dir))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(train_dir))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#basic visualizations\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[\"id_code\"]=train_df[\"id_code\"].apply(lambda x:x+\".png\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Loading two images from every category"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets display one one from each category\nimport cv2\nimport matplotlib.pyplot as plt\nimg = []\nimg.append(os.path.join(train_dir,'002c21358ce6.png'))\nimg.append(os.path.join(train_dir,'005b95c28852.png'))\n\nimg.append(os.path.join(train_dir,'0124dffecf29.png'))\nimg.append(os.path.join(train_dir,'00cb6555d108.png'))\n\nimg.append(os.path.join(train_dir,'03676c71ed1b.png'))\nimg.append(os.path.join(train_dir,'03747397839f.png'))\n\nimg.append(os.path.join(train_dir,'0104b032c141.png'))\nimg.append(os.path.join(train_dir,'03c85870824c.png'))\n\n\nimg.append(os.path.join(train_dir,'03a7f4a5786f.png'))\nimg.append(os.path.join(train_dir,'0318598cfd16.png'))\n\nimages = []\nfor i in range(0,len(img)):\n    images.append(plt.imread(img[i]))    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Some simple visualizations"},{"metadata":{"trusted":true},"cell_type":"code","source":"images[0].shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=[32,32])\ni = 0\nfor img_name in images:\n    plt.subplot(5, 2,i+1)\n    plt.imshow(img_name)\n    if(i<2):\n        plt.title(\"No DR\")\n    elif(i>=2 and i<4):\n        plt.title(\"Mild\")\n    elif(i>=4 and i<6):\n        plt.title(\"Moderate\")\n    elif(i>=6 and i<8):\n        plt.title(\"Severe\")\n    elif(i>=8 and i<10):\n        plt.title(\"Proliferative DR\")\n    i+=1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Data image augmentation"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(rescale = 1/255.,\n                                  horizontal_flip = True,\n                                  width_shift_range = 0.2,\n                                  height_shift_range = 0.2,\n                                  fill_mode = 'nearest',\n                                  validation_split = 0.15,\n                                  zoom_range = 0.3,\n                                  rotation_range = 30)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['diagnosis'] = train_df['diagnosis'].astype('str')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_dir,\n    validation_split = 0.2,\n    x_col = 'id_code',\n    y_col = 'diagnosis',\n    target_size = (800,800),\n    class_mode = 'categorical',\n    batch_size = 32,\n    subset = 'training'\n)\n\nval_generator = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'id_code',\n    y_col = 'diagnosis',\n    directory = train_dir,\n    class_mode = \"categorical\",\n    batch_size = 32,\n    target_size = (800,800),\n    subset = \"validation\"\n    )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We make the first layer non-trainable and add a few dense layers along with Flatten() and a Dropout layer to prevent overfitting"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.models import Sequential, Model\nfrom keras.layers import Dense, Flatten, Activation, Dropout, GlobalAveragePooling2D\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import optimizers, applications\nfrom keras.optimizers import Adam\nfrom keras.callbacks import ModelCheckpoint, LearningRateScheduler, TensorBoard, EarlyStopping\nfrom IPython.display import Image\nfrom keras.preprocessing import image\nfrom keras import optimizers\nfrom keras import layers,models\nfrom keras.applications.imagenet_utils import preprocess_input\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom keras import regularizers\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import DenseNet121, DenseNet169, DenseNet201\nfrom keras.models import Sequential, Model\nfrom keras.layers import Input, Conv2D, MaxPool2D, Dense, Dropout, Activation, Flatten\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.callbacks import ReduceLROnPlateau\nfrom keras.optimizers import Adam\n\nmodel = Sequential()\nmodel.add(Conv2D(32,(3,3),activation = 'relu',input_shape = (800,800,3)))\nmodel.add(Conv2D(32,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\nmodel.add(Conv2D(64,(3,3),activation = 'relu'))\nmodel.add(Conv2D(64,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation = 'relu'))\nmodel.add(MaxPool2D(2,2))\n\nmodel.add(Flatten())\nmodel.add(Dense(512,activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(5,activation = 'softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Implementing callbacks"},{"metadata":{"trusted":true},"cell_type":"code","source":"#some callbacks and tensorboard initialization\n\ncallbacks = [ModelCheckpoint(filepath='best_model.h5', monitor='val_loss', save_best_only=True)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss = 'categorical_crossentropy',optimizer = Adam(),metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Training for 80 epochs!"},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(\n    train_generator,\n    epochs = 80,\n    steps_per_epoch = 20,\n    validation_data = val_generator,\n    validation_steps = 7,\n    callbacks = callbacks\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plotting accuracies and losses\nacc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1,len(acc) + 1)\n\nplt.plot(epochs,acc,'bo',label = 'Training Accuracy')\nplt.plot(epochs,val_acc,'b',label = 'Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs,loss,'bo',label = 'Training loss')\nplt.plot(epochs,val_loss,'b',label = 'Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#make predictions on test images\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\n\nsample_df = pd.read_csv('../input/sample_submission.csv')\n\nsample_df[\"id_code\"]=sample_df[\"id_code\"].apply(lambda x:x+\".png\")\n\ntest_generator = test_datagen.flow_from_dataframe(  \n        dataframe=sample_df,\n        directory = test_dir,    \n        x_col=\"id_code\",\n        target_size = (800,800),\n        batch_size = 1,\n        shuffle = False,\n        class_mode = None\n        )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = model.predict_generator(\n    test_generator,\n    steps=len(test_generator.filenames)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#submission formatting\nfilenames= test_generator.filenames\nresults=pd.DataFrame({\"id_code\":filenames,\n                      \"diagnosis\":np.argmax(preds,axis = 1)})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])\nresults.to_csv(\"submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Go ahead and submit!"},{"metadata":{"trusted":true},"cell_type":"code","source":"count = 0\nfor i in range(0,len(results['diagnosis'])):\n    if(results['diagnosis'][i] == 4):\n        count+=1\n    \ncount","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}