{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nimport numpy as np # linear algebra\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from keras.utils.np_utils import to_categorical # convert to one-hot-encoding\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import LearningRateScheduler","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9769a790a92acc6b918a58747f34325c6382c300"},"cell_type":"code","source":"train_file = \"../input/train.csv\"\ntest_file = \"../input/test.csv\"\noutput_file = \"submission.csv\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"897abd9b0069771e56e1c9a724bf7dca01b55629"},"cell_type":"code","source":"raw_data = np.loadtxt(train_file, skiprows=1, dtype='int', delimiter=',')\nx_train, x_val, y_train, y_val = train_test_split(\n    raw_data[:,1:], raw_data[:,0], test_size=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"501b723ee4095fbc019a043db5e1878c55af18c9"},"cell_type":"code","source":"fig, ax = plt.subplots(2, 1, figsize=(12,6))\nax[0].plot(x_train[0])\nax[0].set_title('784x1 data')\nax[1].imshow(x_train[0].reshape(28,28), cmap='gray')\nax[1].set_title('28x28 data')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d71abf4aa694aaa6eca83c6ed45a02d444675b52"},"cell_type":"code","source":"x_train = x_train.reshape(-1, 28, 28, 1)\nx_val = x_val.reshape(-1, 28, 28, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a66ce0771c4191ea50ce39c304cf7d1f8ee8346"},"cell_type":"code","source":"x_train = x_train.astype(\"float32\")/255.\nx_val = x_val.astype(\"float32\")/255.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9aebc6f20a41ff3fa2c61b1df4122965e08e1819"},"cell_type":"code","source":"y_train[3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b6a988d3557d3c0400c0b35bcccb8fb29da12b5"},"cell_type":"code","source":"y_train = to_categorical(y_train)\ny_val = to_categorical(y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"906b6f72af545bef77ea030408828637fde0796b"},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(filters = 16, kernel_size = (3, 3), activation='relu',\n                 input_shape = (28, 28, 1)))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(filters = 16, kernel_size = (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\n#model.add(Conv2D(filters = 16, kernel_size = (3, 3), activation='relu'))\n#model.add(BatchNormalization())\nmodel.add(MaxPool2D(strides=(2,2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(filters = 32, kernel_size = (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(filters = 32, kernel_size = (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\n#model.add(Conv2D(filters = 32, kernel_size = (3, 3), activation='relu'))\n#model.add(BatchNormalization())\nmodel.add(MaxPool2D(strides=(2,2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation='relu'))\nmodel.add(Dropout(0.25))\nmodel.add(Dense(1024, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(10, activation='softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"302a5a627b6f029696fbe790f73843777d8ece3b"},"cell_type":"code","source":"datagen = ImageDataGenerator(zoom_range = 0.1,\n                            height_shift_range = 0.1,\n                            width_shift_range = 0.1,\n                            rotation_range = 10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b030c6ac49d65aa54f5a970a025e9e903959c97"},"cell_type":"code","source":"model.compile(loss='categorical_crossentropy', optimizer = Adam(lr=1e-4), metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7521d99ee8a9ec5e64109257bea2626f3840081a"},"cell_type":"code","source":"annealer = LearningRateScheduler(lambda x: 1e-3 * 0.9 ** x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d430cd9e5980ce760c88c71985b575ebc598499"},"cell_type":"code","source":"hist = model.fit_generator(datagen.flow(x_train, y_train, batch_size=16),\n                           steps_per_epoch=500,\n                           epochs=20, #Increase this when not on Kaggle kernel\n                           verbose=2,  #1 for ETA, 0 for silent\n                           validation_data=(x_val[:400,:], y_val[:400,:]), #For speed\n                           callbacks=[annealer])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"09877e848f1201559f2cfea624079e1555e23ab5"},"cell_type":"code","source":"final_loss, final_acc = model.evaluate(x_val, y_val, verbose=0)\nprint(\"Final loss: {0:.4f}, final accuracy: {1:.4f}\".format(final_loss, final_acc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8b7c77c7422b823e56e0475e363d9ba041e4b2d"},"cell_type":"code","source":"y_hat = model.predict(x_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07532104051c06360fc3cb3d7f196eda2d08c2a6"},"cell_type":"code","source":"y_pred = np.argmax(y_hat, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"309c080e2b6b13a4f43192665318feee0066edac"},"cell_type":"code","source":"y_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f18a1490e34965f8bb02fa13e7d37e38638ff4cd"},"cell_type":"code","source":"mnist_testset = np.loadtxt(test_file, skiprows=1, dtype='int', delimiter=',')\nx_test = mnist_testset.astype(\"float32\")\nx_test = x_test.reshape(-1, 28, 28, 1)/255.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e809ccad0a1fe630a055caee275162b3a5fc98aa"},"cell_type":"code","source":"y_hat = model.predict(x_test, batch_size=64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"893b944b5d263045d508f2ffe529aad4462ea24d"},"cell_type":"code","source":"y_pred = np.argmax(y_hat,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b3759872d2f1f7c3b000de2eab9e235fdab982a"},"cell_type":"code","source":"with open(output_file, 'w') as f :\n    f.write('ImageId,Label\\n')\n    for i in range(len(y_pred)) :\n        f.write(\"\".join([str(i+1),',',str(y_pred[i]),'\\n']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4df6af9b5360f7a087e2f4a6c8558b879f073b8f"},"cell_type":"code","source":"df=pd.DataFrame(y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b8f269aeea18b90523acd5e2221c3702c3085b8"},"cell_type":"code","source":"df.to_csv('sub.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"531e2274c2832bf719aad12f9ef394224ce48e9d"},"cell_type":"code","source":"a=pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84b9b3aed194832f5b7c748bc8b6510068f89225"},"cell_type":"code","source":"results = pd.Series(y_pred,name=\"Label\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3ba67d1c031a5d8f0921a35e040587e67ae9cf8"},"cell_type":"code","source":"submission = pd.concat([pd.Series(range(1,28001),name = \"ImageId\"),results],axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c5cd9c48f22cc936eaabb23dc13bbc4e8e92f64"},"cell_type":"code","source":"submission.to_csv(\"cnn_mnist_datagen.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c0856a7f1e96bc320906adf92747eb17b847059"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}