{"cells":[{"metadata":{"_uuid":"34a505ed349403d0244805cc027cf5b4a196bae4"},"cell_type":"markdown","source":"<h1><center> CNN with Keras </center><h1>"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport matplotlib.pyplot as plt\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n\nfrom time import time\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.vgg16 import VGG16\n\nfrom keras.callbacks import TensorBoard\n\n# import the necessary packages\nfrom keras.models import Sequential\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras import backend as K","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"510e75653d766d32142ae0cc3324863e8d8834da"},"cell_type":"markdown","source":"## Some Exploratory Data Analysis (EDA)"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/train_labels.csv')\nprint(df.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c226c44aa0a56dea6249542d9046b784fdedf47"},"cell_type":"code","source":"print('Number of image : ', len(df))\nprint('Ratio labels : ', sum(df['label'].values)/len(df))\nimg = plt.imread(\"../input/train/\"+df.iloc[0]['id']+'.tif')\nprint('Images shape', img.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07490ccf374d58927d0c8a0e528387979d38aac4"},"cell_type":"code","source":"for i in range(5):\n    img = plt.imread(\"../input/train/\"+df.iloc[i]['id']+'.tif')\n    print(df.iloc[i]['label'])\n    plt.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"90efa6916403cb5a7446639bc576e461e460bdbc"},"cell_type":"markdown","source":"## Dataset Generators"},{"metadata":{"trusted":true,"_uuid":"6caeaa84fb97c9160c98ca960d083b7eafca5f16"},"cell_type":"code","source":"df = pd.read_csv('../input/train_labels.csv')\n\ntrain_datagen = ImageDataGenerator(\n       # horizontal_flip=True,\n       #vertical_flip=True,\n       #brightness_range=[0.5, 1.5],\n       #fill_mode='reflect',                               \n        #rotation_range=15,\n        rescale=1./255,\n        #shear_range=0.2,\n        #zoom_range=0.2\n        validation_split=0.15\n    \n)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_path = '../input/train'\nvalid_path = '../input/train'\n\ntrain_generator = train_datagen.flow_from_dataframe(\n                dataframe=df,\n                directory=train_path,\n                x_col = 'id',\n                y_col = 'label',\n                has_ext=False,\n                subset='training',\n                target_size=(96, 96),\n                batch_size=64,\n                class_mode='binary'\n                )\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n                dataframe=df,\n                directory=valid_path,\n                x_col = 'id',\n                y_col = 'label',\n                has_ext=False,\n                subset='validation', # This is the trick to properly separate train and validation dataset\n                target_size=(96, 96),\n                batch_size=64,\n                shuffle=False,\n                class_mode='binary'\n                )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6c52cbdd1a74085b164c6f328d1eefc35287968d"},"cell_type":"markdown","source":"## Model definition "},{"metadata":{"trusted":true,"_uuid":"fe53c63f060d2eb6ee8c9d4dd5ef8a0de4288f5d"},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(filters = 16, kernel_size = 3, padding = 'same', activation = 'relu', input_shape = (96, 96, 3)))\nmodel.add(Conv2D(filters = 16, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 16, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size = 3))\n\nmodel.add(Conv2D(filters = 32, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size = 3))\n\nmodel.add(Conv2D(filters = 64, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = 3, padding = 'same', activation = 'relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size = 3))\n\nmodel.add(Conv2D(filters = 128, kernel_size = 3, padding = 'same', activation = 'elu'))\nmodel.add(Conv2D(filters = 128, kernel_size = 3, padding = 'same', activation = 'elu'))\nmodel.add(Conv2D(filters = 128, kernel_size = 3, padding = 'same', activation = 'elu'))\nmodel.add(Dropout(0.3))\n\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(1, activation = 'sigmoid'))\nmodel.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1d877d3ab220d734dd72e36e28a969004a84ebcf"},"cell_type":"markdown","source":"## Training routine "},{"metadata":{"trusted":true,"_uuid":"08bac7a2269ee68afcad1e9b451d89f298038c64"},"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\nSTEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=validation_generator.n//validation_generator.batch_size\n\nmodel.fit_generator(\n                train_generator,\n                steps_per_epoch=STEP_SIZE_TRAIN,\n                epochs=15,\n                validation_data=validation_generator,\n                validation_steps=STEP_SIZE_VALID)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c15a464419e0cb98f94aeea03aa3433435ad370"},"cell_type":"markdown","source":"## Submission"},{"metadata":{"trusted":true,"_uuid":"25bd91c2c3f2c157d5c7eef968392a19050fa39e"},"cell_type":"code","source":"test_df = pd.read_csv('../input/sample_submission.csv')\n\nfrom matplotlib.pyplot import imread\n# Kaggle testing\nfrom glob import glob\nTESTING_BATCH_SIZE = 64\ntesting_files = glob(os.path.join('../input/test/','*.tif'))\nsubmission = pd.DataFrame()\nprint(len(testing_files))\nfor index in range(0, len(testing_files), TESTING_BATCH_SIZE):\n    data_frame = pd.DataFrame({'path': testing_files[index:index+TESTING_BATCH_SIZE]})\n    data_frame['id'] = data_frame.path.map(lambda x: x.split('/')[3].split(\".\")[0])\n    data_frame['image'] = data_frame['path'].map(imread)\n    images = np.stack(data_frame.image, axis=0)\n    predicted_labels = [model.predict(np.expand_dims(image/255.0, axis=0))[0][0] for image in images]\n    predictions = np.array(predicted_labels)\n    data_frame['label'] = predictions\n    submission = pd.concat([submission, data_frame[[\"id\", \"label\"]]])\n    if index % 1000 == 0 :\n        print(index/len(testing_files) * 100)\nsubmission.to_csv('submission_new_model.csv', index=False, header=True)\nprint(submission.head())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}