{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train=pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8335a5bb3ea1a89ce252948a66a8ac3c6559adc4"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bf4230326b39e882157211295f3dcc9458379b1"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2ee36efef73ad6512b0a570fc3602c8a4b7b02c"},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d909902ac1ef76faad6baa1e6e613c7809900cd2"},"cell_type":"code","source":"# put labels into y_train variable\ny_train = train[\"Id\"]\n# Drop 'Id' column\nX_train = train.drop(labels = [\"Id\"], axis = 1)\ny_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e85beb56e0d242a9569a798e8d146afe579992ee"},"cell_type":"code","source":"from keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\n\ndef prepareImages(train, shape, path):\n    \n    x_train = np.zeros((shape, 100, 100, 3))\n    count = 0\n    \n    for fig in train['Image']:\n        \n        #load images into images of size 100x100x3\n        img = image.load_img(\"../input/\"+path+\"/\"+fig, target_size=(100, 100, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n\n        x_train[count] = x\n        if (count%500 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n        \n    return x_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38e52090867dfc57f82a46c1ab322cd9225984b2"},"cell_type":"code","source":"x_train = prepareImages(train, train.shape[0], \"train\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9deee675d279bbd0aa2c8f075266a7dcff9f857e"},"cell_type":"code","source":"x_train = x_train / 255.0\nprint(\"x_train shape: \",x_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32e50456358b5370427c8bc206d457efcd809dbf"},"cell_type":"code","source":"x_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f3e82fef0c3b32c979b748b4db0a04baffc933d"},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0b03a53f3aebb76ae56cb4d9c33379ae93f198b"},"cell_type":"code","source":"y_train = label_encoder.fit_transform(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"816a672192781b5f392559d0d724640ab08ad973"},"cell_type":"code","source":"from keras.utils.np_utils import to_categorical\ny_train = to_categorical(y_train, num_classes = 5005)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ec199378a5e46054ca3b651e542c0bebdf05652"},"cell_type":"code","source":"y_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e34cd8780d27a87d2fffc46b301f2afae537193"},"cell_type":"code","source":"#start building the model\n\nfrom keras.layers import Activation, BatchNormalization\nfrom keras.layers import MaxPooling2D, Dropout\nfrom keras.models import Sequential\nfrom keras.layers import Flatten, Dense\nfrom keras.layers import Conv2D\nfrom keras.optimizers import Adam\n\nmodel2 = Sequential()\nmodel2.add(Conv2D(32, (5,5), input_shape = (x_train.shape[1:]), padding = 'same'))\nmodel2.add(Activation('relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPooling2D(pool_size =  (2,2), strides = (2,2)))\nmodel2.add(Dropout(0.2))\n\nmodel2.add(Conv2D(32, (3,3), padding = 'same'))\nmodel2.add(Activation('relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPooling2D(pool_size =  (2,2), strides = (2,2)))\nmodel2.add(Dropout(0.2))\n\n# model1.add(Conv2D(128, (3,3), padding = 'same'))\n# model1.add(Activation('relu'))\n# model1.add(BatchNormalization())\n# model1.add(MaxPooling2D(pool_size =  (2,2)))\n\nmodel2.add(Flatten())\n\nmodel2.add(Dense(128))\nmodel2.add(Activation('relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(Dropout(0.5))\n\nmodel2.add(Dense(y_train.shape[1]))\nmodel2.add(Activation('softmax'))\n\nmodel2.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7d2e9e0e578ba088d9c0a4ab7235c832c10950b"},"cell_type":"code","source":"#compile the model\noptim = Adam(lr = 0.001) #using the already available learning rate scheduler\nmodel2.compile(loss = 'categorical_crossentropy', optimizer = optim, metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b81a3e1cabf24e5e4a832e06fef6e8bcad790c9"},"cell_type":"code","source":"#fit the model on our dataset\nhistory1 = model2.fit(x_train, y_train, epochs = 10, batch_size = 64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3eff1aef04abae68c0dee75a7cd9114e8835db9"},"cell_type":"code","source":"test_data = os.listdir(\"../input/test/\")\nprint(len(test_data))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dde2952e152e19d76800a66d8071e1c7ffb7a5de"},"cell_type":"code","source":"test_data = pd.DataFrame(test_data, columns = ['Image'])\ntest_data['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"2be6234f3269acc46fb6535805ca264cd73ef950"},"cell_type":"code","source":"x_test = prepareImages(test_data, test_data.shape[0], \"test\")\nx_test = x_test.astype('float32') / 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ae59aa34abd2e9c2543ad3c37917a560bc8829d"},"cell_type":"code","source":"predictions = model2.predict(np.array(x_test), verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96dab7370563b924dea640d1c683f542c9896ae5"},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nfor i, pred in enumerate(predictions):\n    test_data.loc[i, 'Id'] = ' '.join(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"680a9338aa454c3ef3188a34f2a0245d7453e018"},"cell_type":"code","source":"test_data.to_csv('model_submission4.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f05ae09f4d4dc1ad129eee0fc21b7c31d13e0cd3"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}