{"cells":[{"metadata":{"_cell_guid":"d9d40dc8-eae2-5e9d-87cf-f15e58f82c29","_uuid":"661668666d5a1b6d61449bc4ed46e5251efb3b67","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"963016ae-a336-dc37-7d91-c41503710698","_uuid":"5cf836dfba26e4fb612516d02b2b30ece28940cd","trusted":true},"cell_type":"code","source":"trainLabels = pd.read_csv(\"../input/trainLabels.csv\")\ntrainLabels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f46fcc9c100290195c6b905558ec59b131652732"},"cell_type":"code","source":"trainLabels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04cea3c62275a3649420a57763ce8c58ffa5ff7e"},"cell_type":"code","source":"print(sum(trainLabels['image'].isnull()),sum(trainLabels['level'].isnull()))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5e72b0dc-412a-da1b-3b75-56811c08b05e","_uuid":"2094eef624c24421be8819c628a666209dbe54d7","trusted":true},"cell_type":"code","source":"import os\n\nlisting = os.listdir(\"../input\") \nlisting.remove(\"trainLabels.csv\")\nnp.size(listing)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c7587b9-2de9-3225-1b17-ed01819e719b","_uuid":"87d610bfe54506c1157c169289746dde3467a19d","trusted":true},"cell_type":"code","source":"from PIL import Image\n\n# input image dimensions\nimg_rows, img_cols = 200, 200\n\nimmatrix = []\nimlabel = []\n\nfor file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    fileName = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n    im = Image.open(\"../input/\" + file)   \n    img = im.resize((img_rows,img_cols))\n    gray = img.convert('L')\n    immatrix.append(np.array(gray).flatten())\n    ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4ea9a682-76b3-f14f-b78b-bf740ae59624","_uuid":"1e501330697d55caf678324e681098566e8a0276","trusted":true},"cell_type":"code","source":"immatrix = np.asarray(immatrix)\nimlabel = np.asarray(imlabel)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c94fde6e-04a1-2b1d-d969-24c7eec5c774","_uuid":"33bee6e4a609d271765b56757731eb38b79b636a","trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\n\ndata,Label = shuffle(immatrix,imlabel, random_state=2)\ntrain_data = [data,Label]\ntype(train_data)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c83f8a0-06ee-d7b9-2f68-0b8378f6bb6e","_uuid":"56d162c60f7c0c8927b39a9b4ecce76fcd57f6b5","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib\n\nimg=immatrix[167].reshape(img_rows,img_cols)\nplt.imshow(img)\nplt.imshow(img,cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"81a382eb-b5ba-b941-0b2a-2790252a907e","_uuid":"ec04a1cad11f5052d097464706cc572f6eedb8d0","trusted":true},"cell_type":"code","source":"#batch_size to train\nbatch_size = 32\n# number of output classes\nnb_classes = 5\n# number of epochs to train\nnb_epoch = 5","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0c59ad38-3144-807b-e5d2-7e56974162d2","_uuid":"dbc03b4812745d5133c48b4ca372cc545cd03cb1","trusted":true},"cell_type":"code","source":"# number of convolutional filters to use\nnb_filters = 32\n# size of pooling area for max pooling\nnb_pool = 2\n# convolution kernel size\nnb_conv = 3","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8b52b20b-0e5e-91cb-664d-a33167995111","_uuid":"dff8349af7e9222738a0c93a21462581828f147e","trusted":true},"cell_type":"code","source":"(X, y) = (train_data[0],train_data[1])\nfrom sklearn.cross_validation import train_test_split\n\n# STEP 1: split X and y into training and testing sets\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=4)\n\nprint(X_train.shape)\nprint(X_test.shape)\n\n#X_train = X_train.reshape(X_train.shape[0], 1, img_rows, img_cols)\n#X_test = X_test.reshape(X_test.shape[0], 1, img_rows, img_cols)\n\nX_train = X_train.reshape(X_train.shape[0], img_cols, img_rows, 1)\nX_test = X_test.reshape(X_test.shape[0], img_cols, img_rows, 1)\n\nX_train = X_train.astype('float32')\nX_test = X_test.astype('float32')\n\nX_train /= 255\nX_test /= 255\n\nprint('X_train shape:', X_train.shape)\nprint(X_train.shape[0], 'train samples')\nprint(X_test.shape[0], 'test samples')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6c5d9169-cdb8-df65-2ea3-a3a6f9694fa3","_uuid":"3548ca46f4f34ff89eb5b02884710acf3f3c21bf","trusted":true},"cell_type":"code","source":"from keras.utils import np_utils\n\n# convert class vectors to binary class matrices\nY_train = np_utils.to_categorical(y_train, nb_classes)\nY_test = np_utils.to_categorical(y_test, nb_classes)\n\ni = 100\nplt.imshow(X_train[i, 0], interpolation='nearest')\nprint(\"label : \", Y_train[i,:])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c1c4b87a-053e-c580-fcf6-93d80d5ed8e6","_uuid":"ee79198316fbadd013c9786576290ea39024e637","trusted":true},"cell_type":"code","source":"\nfrom keras.models import Sequential\nfrom keras.layers.core import Dense, Dropout, Activation, Flatten\nfrom keras.layers.convolutional import Convolution2D, MaxPooling2D,AveragePooling2D\nfrom keras.optimizers import SGD,RMSprop,adam\n\nmodel = Sequential()\n\nmodel.add(Convolution2D(nb_filters, nb_conv, nb_conv,\n                        border_mode='valid',\n                        input_shape=(img_cols, img_rows, 1)))\nconvout1 = Activation('relu')\nmodel.add(convout1)\nmodel.add(Convolution2D(nb_filters, nb_conv, nb_conv))\nconvout2 = Activation('relu')\nmodel.add(convout2)\nmodel.add(AveragePooling2D(pool_size=(nb_pool, nb_pool)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(128))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(nb_classes))\nmodel.add(Activation('softmax'))\nmodel.compile(loss='categorical_crossentropy', optimizer='adadelta')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e7871692-22bd-eceb-d429-deac0787fa09","_uuid":"eb2ef63a3c62d4b0d95a16961048f2be6298a252","trusted":true,"scrolled":true},"cell_type":"code","source":"from IPython.display import SVG\nfrom keras.utils.vis_utils import model_to_dot\n\nSVG(model_to_dot(model).create(prog='dot', format='svg'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"19b57223-dcd9-c3f0-35b3-e693e6fac730","_uuid":"da026673ac1729ec32e5e26bc895907531910eae","trusted":true},"cell_type":"code","source":"from pydot import pydot.Graph\nimport graphviz\nfrom keras.utils import plot_model\nplot_model(model, to_file='model.png')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"85aa895e-a687-3874-e965-89dc9ff00462","_uuid":"2d3a352fa9776f30524de5367ad01e971c69a864","trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\n# create generators  - training data will be augmented images\nvalidationdatagenerator = ImageDataGenerator()\ntraindatagenerator = ImageDataGenerator(width_shift_range=0.1,height_shift_range=0.1,rotation_range=15,zoom_range=0.1 )\n\nbatchsize=8\ntrain_generator=traindatagenerator.flow(X_train, Y_train, batch_size=batchsize) \nvalidation_generator=validationdatagenerator.flow(X_test, Y_test,batch_size=batchsize)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"633b8a59925cf1704fb47c694b328f364439468d"},"cell_type":"code","source":"traindatagenerator","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"77510f9e-4f66-e38d-3620-a7b33cdd04d1","_uuid":"4a3cdb13519f21cae71301f459c092f876196042","trusted":true},"cell_type":"code","source":"#hist = model.fit(X_train, Y_train, batch_size=batch_size, nb_epoch=nb_epoch, verbose=1, validation_data=(X_test, Y_test))\n\nmodel.fit_generator(train_generator, steps_per_epoch=int(len(X_train)/batchsize), epochs=4, validation_data=validation_generator, validation_steps=int(len(X_test)/batchsize))\n\n#hist = model.fit(X_train, Y_train, batch_size=batch_size, nb_epoch=nb_epoch,\n#              show_accuracy=True, verbose=1, validation_split=0.2)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"908a049d-0c6b-a2b9-8b4d-3df740b525b8","_uuid":"149c17ba17a4cd008495b71a21a9db79273523e1","trusted":true},"cell_type":"code","source":"score = model.evaluate(X_test, Y_test, verbose=0)\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9edd5723-70ee-223c-2be0-feb69d4a8843","_uuid":"7330e17486ba1263d30c68be7350191cc5a251ff","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba9de95f887b2de9e227152907d65cab7f50e41e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}