{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"04f969e2-507f-4c70-e23d-f0bfaa39afd6"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a8c2da12-99b4-5bd3-8062-a77e5e43d8ec"},"outputs":[],"source":"trainLabels = pd.read_csv(\"../input/trainLabels.csv\")\ntrainLabels.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"c1b3e769-27db-27a5-8706-77ef05706e7c"},"outputs":[],"source":"import os\n\nlisting = os.listdir(\"../input\") \nlisting.remove(\"trainLabels.csv\")\nnp.size(listing)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"7edc5208-d3c8-367e-4bc1-1e7903dc1dc1"},"outputs":[],"source":"from PIL import Image\nfrom keras.applications.vgg16 import preprocess_input\nfrom keras.preprocessing import image\n\n# input image dimensions\nimg_rows, img_cols = 224, 224\n\nimmatrix = []\nimlabel = []\n\nfor file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    fileName = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n    im = Image.open(\"../input/\" + file)\n    img = im.resize((img_rows,img_cols))\n    #img4d = np.expand_dims(img, axis=0)\n    #img4d = preprocess_input(img4d)\n    immatrix.append(np.array(img))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"c1651873-0308-c61c-a019-23cfc1f6cd01"},"outputs":[],"source":"immatrix = np.asarray(immatrix)\nimlabel = np.asarray(imlabel)\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"473978c3-6f32-97db-5e19-3b8d8a748deb"},"outputs":[],"source":"(X, y) = (train_data[0],train_data[1])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"aa553a72-8ebe-4b0a-0d68-4bd474a5502d"},"outputs":[],"source":"from sklearn.utils import shuffle\ndata,Label = shuffle(immatrix,imlabel, random_state=2)\ntrain_data = [data,Label]\ntype(train_data)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"eeddde53-1387-7fcd-4f6f-8f61765396a0"},"outputs":[],"source":"from sklearn.cross_validation import train_test_split\n\n# STEP 1: split X and y into training and testing sets\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=4)\n\nprint(X_train.shape)\nprint(X_test.shape)\n\nX_train = X_train.astype('float32')\nX_test = X_test.astype('float32')\n\nX_train /= 255\nX_test /= 255\n\nprint('X_train shape:', X_train.shape)\nprint(X_train.shape[0], 'train samples')\nprint(X_test.shape[0], 'test samples')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"bc5b8d4e-ba0c-5984-2f02-e43d5bf617fd"},"outputs":[],"source":""},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"59d4f867-e8cb-d4c4-3ed4-f0ffe7f72c44"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}