{"cells":[{"metadata":{"_cell_guid":"d9d40dc8-eae2-5e9d-87cf-f15e58f82c29","_uuid":"12e04b164c8d699edda5c144a67c7c1a3a4bc7d9","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"963016ae-a336-dc37-7d91-c41503710698","_uuid":"c3a6c949653dd0c6fcc51919efc518148aacf450","trusted":true},"cell_type":"code","source":"trainLabels = pd.read_csv(\"../input/trainLabels.csv\")\ntrainLabels.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5e72b0dc-412a-da1b-3b75-56811c08b05e","_uuid":"59e04c5e5ada400c80ca308010d7cf49738771d4","trusted":true},"cell_type":"code","source":"import os\nlisting = os.listdir(\"../input\") \nlisting.remove(\"trainLabels.csv\")\nnp.size(listing)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c7587b9-2de9-3225-1b17-ed01819e719b","_uuid":"01df70976cdbfe9926d468a956d24a3c5f78796f","trusted":true},"cell_type":"code","source":"from PIL import Image\n# input image dimensions\nimg_rows, img_cols = 200, 200\nimmatrix = []\nimlabel = []\nfor file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    fileName = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n    im = Image.open(\"../input/\" + file)   \n    img = im.resize((img_rows,img_cols))\n    gray = img.convert('L')\n    immatrix.append(np.array(gray).flatten())","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4ea9a682-76b3-f14f-b78b-bf740ae59624","_uuid":"34d145d303d3fbc11e87693b67eecb26fea1f00c","trusted":true},"cell_type":"code","source":"immatrix = np.asarray(immatrix)\nimlabel = np.asarray(imlabel)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c94fde6e-04a1-2b1d-d969-24c7eec5c774","_uuid":"344dea6e433b8c6807121755958e14aca003b45e","trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\ndata,Label = shuffle(immatrix,imlabel, random_state=2)\ntrain_data = [data,Label]\ntype(train_data)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c83f8a0-06ee-d7b9-2f68-0b8378f6bb6e","_uuid":"dfb2edc910c3ff620e1b76f642876d9fcd892914","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib\nimg=immatrix[167].reshape(img_rows,img_cols)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"81a382eb-b5ba-b941-0b2a-2790252a907e","_uuid":"edaacdd20bd53058020f2853a09234d6757ff3a5","trusted":true},"cell_type":"code","source":"#batch_size to train\nbatch_size = 32\n# number of output classes\nnb_classes = 5\n# number of epochs to train\nnb_epoch = 5","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0c59ad38-3144-807b-e5d2-7e56974162d2","_uuid":"4e101f1f6973ea6b2a2d3da82c29bbc4164a0140","trusted":true},"cell_type":"code","source":"# number of convolutional filters to use\nnb_filters = 32\n# size of pooling area for max pooling\nnb_pool = 2\n# convolution kernel size\nnb_conv = 3","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8b52b20b-0e5e-91cb-664d-a33167995111","_uuid":"a4a2f6fb92d57c086fbe5646cd43d83d6832e4f4","trusted":true},"cell_type":"code","source":"(X, y) = (train_data[0],train_data[1])\nfrom sklearn.model_selection import train_test_split\n# STEP 1: split X and y into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=4)\nprint(X_train.shape)\nprint(X_test.shape)\n ##reshapping data and normalize it\nX_train = X_train.reshape(X_train.shape[0], img_cols, img_rows, 1)\nX_test = X_test.reshape(X_test.shape[0], img_cols, img_rows, 1)\nX_train = X_train.astype('float32')\nX_test = X_test.astype('float32')\n#normalize\nX_train /= 255\nX_test /= 255\nprint('X_train shape:', X_train.shape)\nprint(X_train.shape[0], 'train samples')\nprint(X_test.shape[0], 'test samples')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6c5d9169-cdb8-df65-2ea3-a3a6f9694fa3","_uuid":"4ffce8afb6e3e96bf297377fd279a03c063efff7","trusted":true},"cell_type":"code","source":"from keras.utils import np_utils\n# convert class vectors to binary class matrices\n#convert to categorical\nY_train = np_utils.to_categorical(y_train, nb_classes)\nY_test = np_utils.to_categorical(y_test, nb_classes)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c1c4b87a-053e-c580-fcf6-93d80d5ed8e6","_uuid":"6290d3bb1e6818ac7f2d0c6c71eac867f05052ad","trusted":true},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers.core import Dense, Dropout, Activation, Flatten\nfrom keras.layers.convolutional import Convolution2D, MaxPooling2D\nfrom keras.optimizers import SGD,RMSprop,adam\n\nbase_model = Sequential()\n\nbase_model.add(Convolution2D(64,3,3,border_mode='valid',input_shape=(img_cols, img_rows, 1),activation=\"relu\"))\nbase_model.add(MaxPooling2D(pool_size=(2,2)))\n  \nbase_model.add(Convolution2D(32,3,3,activation=\"relu\"))\nbase_model.add(MaxPooling2D(pool_size=(2,2)))\n\nbase_model.add(Convolution2D(16,3,3,activation=\"relu\"))\nbase_model.add(MaxPooling2D(pool_size=(2,2)))\n\nbase_model.add(Flatten())\nbase_model.add(Dense(128,activation=\"relu\"))\nbase_model.add(Dense(5,activation=\"softmax\")) \nbase_model.compile(loss='categorical_crossentropy', optimizer='SGD',metrics=[\"accuracy\"])\nbase_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d9efbbfba259b5d250be6b592aa8885411aeb24"},"cell_type":"code","source":"hist = base_model.fit(X_train, Y_train, batch_size=64, nb_epoch=5, verbose=1,validation_split=0.2,shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9efffb5419cb90a5a2bb2e754adf7c8821a43ab0"},"cell_type":"code","source":"history_dict=hist.history\nhistory_dict.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d91566795c193d1b02045be5e33cb83b132a2ce"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nloss_values = history_dict['loss']\nval_loss_values = history_dict['val_loss']\nepochs = range(1, len(history_dict['acc']) + 1)\nplt.plot(epochs, loss_values, 'bo', label='Training loss')\nplt.plot(epochs, val_loss_values, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8aa89024ed2317efda5dcd879f8a10a6e1af0749"},"cell_type":"code","source":"plt.clf()\nacc_values = history_dict['acc']\nval_acc_values = history_dict['val_acc']\nplt.plot(epochs, history_dict['acc'], 'bo', label='Training acc')\nplt.plot(epochs, history_dict['val_acc'], 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"534b625926a577b67ed19cdcd3dac32a1930de2b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5dabc0e9c0cb83b68d31abd9552bf0dd4079d24"},"cell_type":"code","source":"score = base_model.evaluate(X_test, Y_test, verbose=0)\nprint(\"loss= \",score[0],\"acc= \",score[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14190c4e0be9d49de3ab7e5b5a3de7310ca998a4"},"cell_type":"code","source":"#let's add a drop out layers && adam optimizer\nfrom keras.models import Sequential\nfrom keras.layers.core import Dense, Dropout, Activation, Flatten\nfrom keras.layers.convolutional import Convolution2D, MaxPooling2D\nfrom keras.optimizers import SGD,RMSprop,adam\n\ndrop_model = Sequential()\n\ndrop_model.add(Convolution2D(64,3,3,border_mode='valid',input_shape=(img_cols, img_rows, 1),activation=\"relu\"))\ndrop_model.add(MaxPooling2D(pool_size=(2,2)))\ndrop_model.add(Dropout(0.5))\ndrop_model.add(Convolution2D(32,3,3,activation=\"relu\"))\ndrop_model.add(MaxPooling2D(pool_size=(2,2)))\ndrop_model.add(Dropout(0.5))\ndrop_model.add(Convolution2D(16,3,3,activation=\"relu\"))\ndrop_model.add(MaxPooling2D(pool_size=(2,2)))\ndrop_model.add(Dropout(0.5))\ndrop_model.add(Flatten())\ndrop_model.add(Dense(128,activation=\"relu\"))\ndrop_model.add(Dense(5,activation=\"softmax\")) \ndrop_model.compile(loss='categorical_crossentropy', optimizer='adam',metrics=[\"accuracy\"])\ndrop_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82dd7d12bb03d746ec5ae6bf3e26202ef713dc28"},"cell_type":"code","source":"hist = drop_model.fit(X_train, Y_train, batch_size=128, nb_epoch=10, verbose=1,validation_split=0.2,shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa04392aa83ff0154dcd9807187c72703009eea3"},"cell_type":"code","source":"history_dict2=hist.history\nhistory_dict2.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab45d438e2cae626480ef493c37718ff337f3379"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nloss_values = history_dict2['loss']\nval_loss_values = history_dict2['val_loss']\nepochs = range(1, len(history_dict2['acc']) + 1)\nplt.plot(epochs, loss_values, 'bo', label='Training loss')\nplt.plot(epochs, val_loss_values, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd616bddbf3d64ed1545b2f6c6a56f5a870017ae"},"cell_type":"code","source":"plt.clf()\nacc_values = history_dict2['acc']\nval_acc_values = history_dict2['val_acc']\nplt.plot(epochs, history_dict2['acc'], 'bo', label='Training acc')\nplt.plot(epochs, history_dict2['val_acc'], 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"85aa895e-a687-3874-e965-89dc9ff00462","_uuid":"3ca8ceeb90002bbce65e37d955793d7989f12b44","trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n# create generators  - training data will be augmented images\nvalidationdatagenerator = ImageDataGenerator()\ntraindatagenerator = ImageDataGenerator(width_shift_range=0.1,height_shift_range=0.1,rotation_range=15,zoom_range=0.1 )\nbatchsize=64\ntrain_generator=traindatagenerator.flow(X_train, Y_train, batch_size=batchsize) \nvalidation_generator=validationdatagenerator.flow(X_test, Y_test,batch_size=batchsize)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"77510f9e-4f66-e38d-3620-a7b33cdd04d1","_uuid":"5fb0f5419d3278813feeee452cb5f2513b024f26","trusted":true},"cell_type":"code","source":"history=base_model.fit_generator(train_generator, steps_per_epoch=int(len(X_train)/batchsize), epochs=3, validation_data=validation_generator, validation_steps=int(len(X_test)/batchsize))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"908a049d-0c6b-a2b9-8b4d-3df740b525b8","_uuid":"458e22fbea3fa56329d75afb1c68df89634bfb2e","trusted":true},"cell_type":"code","source":"score = base_model.evaluate(X_test, Y_test, verbose=0)\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9edd5723-70ee-223c-2be0-feb69d4a8843","_uuid":"526568cf92c22a4c21f20e965edb8221456db005","trusted":true},"cell_type":"code","source":"history=drop_model.fit_generator(train_generator, steps_per_epoch=int(len(X_train)/batchsize), epochs=3, validation_data=validation_generator, validation_steps=int(len(X_test)/batchsize))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecf5f780740aab3cd463c44faed76c0fc2617e54"},"cell_type":"code","source":"score = drop_model.evaluate(X_test, Y_test, verbose=0)\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a962479b1e9f391dcdb68fd88c3e70e5d8b0a020"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64e29c7774e2c841ad4c0d5f073e703a419d768a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5daf1bacaba035a611c9bf0f946f9c82b7db303"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}