{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport seaborn as sns\nimport os\nprint(os.listdir(\"../input\"))\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.preprocessing import image\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\nfrom keras.layers import Flatten\nfrom keras.layers.convolutional import Convolution2D\nfrom keras.layers.convolutional import MaxPooling2D\n\nfrom keras.utils import np_utils\nfrom keras import backend as K\n#K.set_image_dim_ordering('th')\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"904a388859527c1290b5689a4d8c1c2f1a3ae9f8"},"cell_type":"code","source":"#fixing a random seed\nseed = 7\nnp.random.seed(seed)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6262efca279592a6035b18fafb40d13a7cf3c036"},"cell_type":"code","source":"training_data = pd.read_csv(\"../input/train.csv\")\ntraining_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8a609016ec06f77caa8cfb0edacd3026da74f38"},"cell_type":"code","source":"y = training_data['Id']\nX = training_data['Image']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0f8ffaffd2ae5829ba601e145e071432eb164d9"},"cell_type":"code","source":"print(X.shape)\nprint(y.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4026dc05089e0407858e4fc45add13e73d3d6454"},"cell_type":"code","source":"from matplotlib.image import imread\ncount = 0\ntrain_X = np.zeros((X.shape[0], 100, 100, 3))\nprint(train_X.shape)\nfor i in X:\n    \n    #load images into images of size 100x100x3\n    img = image.load_img(\"../input/train/\" + i , target_size=(100, 100, 3))\n    x = image.img_to_array(img)\n    x = preprocess_input(x)\n   # print(x.shape)\n    train_X[count] = x\n    if (count%500 == 0):\n        print(\"Processing image: \", count+1, \", \", i)\n    count += 1\n\nprint(train_X.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e69cbe1047dcbf79b9ace5ac0da6f0915b66c864"},"cell_type":"code","source":"train_X/=255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ae70e4d595947d6eb4c8b7dac055fa5f61bdb30"},"cell_type":"code","source":"testing_data = os.listdir(\"../input/test/\")\nprint(len(testing_data))\n\nn_test = len(testing_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f806288ff2b3d35227a773e67a73a16e56ae597a"},"cell_type":"code","source":"\ncount = 0\ntest_imgs = np.zeros((n_test, 100, 100, 3))\nprint(test_imgs.shape)\nfor i in testing_data:\n    \n    #load images into images of size 100x100x3\n    img = image.load_img(\"../input/test/\" + i , target_size=(100, 100, 3))\n    x = image.img_to_array(img)\n    x = preprocess_input(x)\n   # print(x.shape)\n    test_imgs[count] = x\n    if (count%500 == 0):\n        print(\"Processing image: \", count+1, \", \", i)\n    count += 1\n\nprint(test_imgs.shape)\n\ntest_imgs /= 255\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dae4c1ba54c51879db760c7e41845a59f411f598"},"cell_type":"code","source":"onehot = pd.get_dummies(y)\n#print(onehot)\ntarget_labels = onehot.columns\nprint(target_labels)\ny = onehot.as_matrix()\nprint(y.shape)\nn = y.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce9554cf9dc7063b03c7779b30f568eed310e848"},"cell_type":"code","source":"target_labels[2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35e712c5596972fbe559f0ccaf7b4573e5d53acf"},"cell_type":"code","source":"from PIL import Image\nimg = imread(\"../input/train/659583f73.jpg\" )\nprint(img.shape)\n\nimage = Image.open(\"../input/train/659583f73.jpg\")\nplt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72dfb7b56cc0ca41ff6de8520260f5b2ec0e7f7e"},"cell_type":"code","source":"#Create model\n\nmodel = Sequential()\nmodel.add(Convolution2D(32, (5,5),  input_shape=(100,100,3), activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Convolution2D(32, (5,5), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\nmodel.add(Convolution2D(64, (5,5), activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Convolution2D(64, (5,5), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\nmodel.add(Convolution2D(128, (5,5), activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Convolution2D(128, (5,5), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=(2,2)))\nmodel.add(Flatten())\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1024, activation= 'relu' ))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(512, activation= 'relu' ))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(n, activation= 'softmax' ))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cf4a732046906e4941437b86f50eba2d104fbec"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21a7daaca8b664b93b726c24e188046db6bce447"},"cell_type":"code","source":"\n\nhistory = model.fit(train_X, y, epochs=25, batch_size=64, verbose=1)\n# Final evaluation of the model\n#scores = model.evaluate(X_test, y_test, verbose=0)\n#print(\"Accuracy: %.2f%%\" % (scores[1]*100))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0dbe3f449ab290d9ed37e95f4977b744d46a2d0"},"cell_type":"code","source":"plt.plot(history.history['acc'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f29cd355f10c0b80525df0f8cea4037f4278bae8"},"cell_type":"code","source":"col = ['Image']\ntest_df = pd.DataFrame(testing_data, columns=col)\ntest_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e029552da06932376d9f595caec79e6df770f8fa"},"cell_type":"code","source":"predictions = model.predict(np.array(test_imgs), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e2a94390b0268694ad0fa60afd255faebc2771b"},"cell_type":"code","source":"results = []\nfor i in predictions:\n    results.append(np.argmax(i))\n\nprint(len(results))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67f3485467bcab3620bcc39dbf45d663a610266c"},"cell_type":"code","source":"test_df['Id'] = results\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"718c340349e33a14de624c21eb1424a6a154e801"},"cell_type":"code","source":"final_pred = []\nfor x in test_df.Id:    \n    #print(x)\n    #print(target_labels[x])\n    final_pred.append(target_labels[x])\nprint(len(final_pred))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4fdc738c18b2a9b8d09e848d5d40be0309af5240"},"cell_type":"code","source":"\ntest_df['Id'] = final_pred\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df25fbacd61190077373411b1f85f0a876de6cb4"},"cell_type":"code","source":"test_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb1e9b0ceb16e15a5f3ba1328e2169a99cb2fd64"},"cell_type":"code","source":"test_df.Id.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a1877eb8f0d0d5234f95834b652cafb422fa9bc"},"cell_type":"code","source":"\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}