{"cells":[{"metadata":{"id":"BkqV-XYohPi3","colab_type":"code","outputId":"163ad478-4cc1-4746-f500-0a8331146d76","colab":{"base_uri":"https://localhost:8080/","height":34},"trusted":true,"_uuid":"6567318b77086796b1f2defcb1fb819a735904f6"},"cell_type":"code","source":"#Importing necessary libraries\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport keras","execution_count":null,"outputs":[]},{"metadata":{"id":"cM_Ug6gJhrW9","colab_type":"code","outputId":"bfe49b58-9391-455e-cbcf-f6f35c03bb54","colab":{"base_uri":"https://localhost:8080/","height":122},"trusted":true,"_uuid":"19e1aff073e339f6a7dc83aa59e51e3e283d3d94"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"id":"f1_6Qzqdh6D_","colab_type":"code","outputId":"1b15ce97-0742-46ef-9dda-56368744a165","colab":{"base_uri":"https://localhost:8080/","height":253},"trusted":true,"_uuid":"4b612ac3f46774ab3029abd0e4c3f4dd4814f016"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"rTOf94gki859","colab_type":"code","colab":{},"trusted":true,"_uuid":"fb25079a6faf03bc6c3496f378f75726dd943c50"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"id":"-adFDLk2jBPL","colab_type":"code","outputId":"851ff5dd-202e-4f1e-d1b3-920d02da3fe9","colab":{"base_uri":"https://localhost:8080/","height":34},"trusted":true,"_uuid":"1b7fa3fa1ce198cf67105f8233c0c9eabb2395d9"},"cell_type":"code","source":"train.shape, test.shape","execution_count":null,"outputs":[]},{"metadata":{"id":"Tcdmxi0kjEnP","colab_type":"code","colab":{},"trusted":true,"_uuid":"5cea3f5f633f099d43e8fc35f8a99d746b781167"},"cell_type":"code","source":"#Randomly shuffling the training set (incase consecutive digits are same)\nindices = np.arange(len(train))\nnp.random.shuffle(indices)\ntrainShuffled = train.loc[indices,:]","execution_count":null,"outputs":[]},{"metadata":{"id":"XdzSjbdpkSBN","colab_type":"code","outputId":"e5480116-ea32-4b52-a98a-f85247b9e2a5","colab":{"base_uri":"https://localhost:8080/","height":416},"trusted":true,"_uuid":"8d9f1c21187b0f66778e7cd74347642c48aa6ec4"},"cell_type":"code","source":"#A visual plot of the images\nfor i in range(25):\n  plt.subplot(5,5,i+1)\n  plt.imshow(np.array(trainShuffled.iloc[i,1:]).reshape(28,28), cmap = 'gray')\n  plt.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7cc08a2bcaa90754f8f26297274674efd28b315c"},"cell_type":"markdown","source":"**Convolution Neural Network**\nIt has the following architecture:\nConv2D(32) -> MaxPool -> BN -> Conv2D(64) -> MaxPool -> BN -> Conv2D(128) -> Dropout(50%) -> Dense 10 (output layer)\n\n**Batch Normalization:** it improves the convergence rate in optimization problem, when we are trying to minimize our loss. It has same idea like when we divide the features (input) by 255 (because images are from 0 to 255) or when we use:\nx = (x-mean)/standard deviation\nSo, doing this for all hidden units, sometimes is benefitial.\n\n**Dropout:** it involves randomly turning off certain neurons in the hidden layer it is specified for (0.5 means 50% after every batch, because we are using mini-batch gradient descent). This prevents overfitting, as the network would not rely on certain neurons always to get the class."},{"metadata":{"id":"CZYvp3j8lCUE","colab_type":"code","colab":{},"trusted":true,"_uuid":"063f2c98c86cb5fe5e1bd0425cd515916a15648e"},"cell_type":"code","source":"\n#Creating Keras Model, consiting of Conv2d + Pooling blocks followed by one Dense layer with 10 hidden units\nmodel = keras.models.Sequential()\nmodel.add(keras.layers.Conv2D(32, (3,3), input_shape = (28,28,1,), activation = 'relu', name = 'Conv2D1'))\nmodel.add(keras.layers.MaxPooling2D((2,2)))\nmodel.add(keras.layers.BatchNormalization())\nmodel.add(keras.layers.Conv2D(64, (3,3), activation = 'relu', name = 'Conv2D2'))\nmodel.add(keras.layers.MaxPooling2D(2,2))\nmodel.add(keras.layers.BatchNormalization())\nmodel.add(keras.layers.Conv2D(128, (2,2),activation = 'relu', name = 'Conv2D3'))\nmodel.add(keras.layers.Dropout(0.5))\nmodel.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(10, activation = 'softmax'))","execution_count":null,"outputs":[]},{"metadata":{"id":"Ou5s5n4NlLtL","colab_type":"code","outputId":"2e6d33d9-9d5d-490b-c81a-5f01666d6b65","colab":{"base_uri":"https://localhost:8080/","height":476},"trusted":true,"_uuid":"a6a61377580099baf93c61fcc9ad42d75552ab96"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"535fc93f201bd4eba51c60a1fc6a96a8394a054a"},"cell_type":"markdown","source":"**Data Augmentation**\nIt is another method like Dropout, to prevent overfitting. One way to deal with overfitting is more data. This is not always possible, so what we do here is we create new data from the given data. How? We have images, so we randomly rotate the images by some degrees (not too extreme) and we shear it, or scale the height and width. We can also zoom it. There are many more parameters that can be experiemented with, like flipping the image (which I found not to be useful), but there are more."},{"metadata":{"id":"GqAkYy5_l-hA","colab_type":"code","colab":{},"trusted":true,"_uuid":"0fbafcf4d6a6a8798832e6595464234d5d23212a"},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"id":"Mbam3ZCupuVE","colab_type":"code","colab":{},"trusted":true,"_uuid":"e36ca017d4324cfa720e31f48b31c68217f9d7b4"},"cell_type":"code","source":"generator = ImageDataGenerator(rescale = 1./255, rotation_range = 15, width_shift_range = 0.1, height_shift_range = 0.1,\n                              shear_range = 0.1, zoom_range = 0.1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7730913378b3d3327bf23e58c63ab6be1602187"},"cell_type":"markdown","source":"**Validation or Hold-out Set**\nTo know if our model is overfitting or not, and to have a rough idea about the number of epochs it would require to get our desired result, we use a validation data that is seaparate from the train data and the model hasn't seen this data before.\nIt is somewhat equivalent to test data, but since we sometimes tune the hyperparameters of our network based on results obtained on validation data, we are optimizing our model for validation data and it might not perform that well on test data, but comparison should be somewhat similar."},{"metadata":{"id":"zpFnc3qzqTOL","colab_type":"code","outputId":"4b6500b5-232a-4cc7-a2f9-9aebba100e46","colab":{"base_uri":"https://localhost:8080/","height":34},"trusted":true,"_uuid":"4f5a4eceeb5b227cde1cdce8c468b3729eb62d6e"},"cell_type":"code","source":"#Validation set and train set split\ntrain_X = trainShuffled.iloc[:35000, 1:]\ntrain_y = trainShuffled.iloc[:35000, 0]\ntest_X = trainShuffled.iloc[35000:, 1:]\ntest_y = trainShuffled.iloc[35000:, 0]\ntrain_X.shape, train_y.shape, test_X.shape, test_y.shape","execution_count":null,"outputs":[]},{"metadata":{"id":"4rKk3LMHq6L1","colab_type":"code","colab":{},"trusted":true,"_uuid":"c2b85e5bbf50a170d5c1cb0f61ed42be2c1542b2"},"cell_type":"code","source":"train_X = np.array(train_X).reshape(-1,28,28,1) #Networks expects a 4D input\n                                        #(samples, height ,width, channel), since not RGB channel = 1\ntest_X = np.array(test_X).reshape(-1,28,28,1)","execution_count":null,"outputs":[]},{"metadata":{"id":"BqY0w1W6rMt0","colab_type":"code","outputId":"109d7c17-6d19-481d-f116-206662544bc1","colab":{"base_uri":"https://localhost:8080/","height":34},"trusted":true,"_uuid":"5641c8ae94014e0d539ebba666375b3398b966d7"},"cell_type":"code","source":"train_X.shape, test_X.shape","execution_count":null,"outputs":[]},{"metadata":{"id":"LgVd99N6r2HK","colab_type":"code","colab":{},"trusted":true,"_uuid":"7d21151b4a4fcbd2a7219311fcd02faf02533f7c"},"cell_type":"code","source":"model.compile(optimizer = keras.optimizers.Adam(.001), loss = 'sparse_categorical_crossentropy',\n              metrics = ['accuracy']) #Our y_train is numerical, so instead of categorical_crossentropy\n                                      #we use sparse_categorical_cross_entropy, and a learning rate 0.001","execution_count":null,"outputs":[]},{"metadata":{"id":"Y6DtzkT84TeV","colab_type":"code","colab":{},"trusted":true,"scrolled":true,"_uuid":"4a3a1d98c8945dd07fb70fa923835d75428d6876"},"cell_type":"code","source":"myDataGen = generator.flow(train_X,train_y, batch_size = 100) #We get data augmented images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59ad8a1e80389273fbcffec55c35946992e6b1e3"},"cell_type":"code","source":"test_X = test_X/255. #Also normalize test data, since we did it with train data","execution_count":null,"outputs":[]},{"metadata":{"id":"vRuxXtaJ4vn7","colab_type":"code","outputId":"abaa40cc-773c-4fe1-ad5a-ede49770309b","colab":{"base_uri":"https://localhost:8080/","height":357},"trusted":true,"_uuid":"9d90f29750933bec358e0537bb04a50fd122cc00"},"cell_type":"code","source":"hist  = model.fit_generator(myDataGen, steps_per_epoch = 350, epochs = 40, validation_data = (test_X, test_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b1c71931126de8387b51943ced901f62e4f5710"},"cell_type":"code","source":"hist.history.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6cbdd9da2cdf24cd4e7ecf596db0387ed229db20"},"cell_type":"code","source":"val_loss = np.array(hist.history['val_loss'])\nval_acc = np.array(hist.history['val_acc'])\ntrain_loss = np.array(hist.history['loss'])\ntrain_acc = np.array(hist.history['acc'])\nepochs = len(train_acc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"978be8f651a2eb564ad638854451c8f74dd3c42f"},"cell_type":"code","source":"accuracies = pd.DataFrame(train_acc, columns = ['train_acc'])\naccuracies['val_acc'] = pd.DataFrame(val_acc)\nlosses = pd.DataFrame(train_loss, columns = ['train_loss'])\nlosses['val_loss'] = pd.DataFrame(val_loss)\nlosses['epochs'] = pd.DataFrame(list(range(epochs)))\naccuracies['epochs'] = pd.DataFrame(list(range(epochs)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b9efe7fa3255bd5cce0f5103c91748e251e5257"},"cell_type":"code","source":"accuracies.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6329c7d4248aa75b407a05edacdc467f7dd6bf6f"},"cell_type":"code","source":"\nplt.scatter(x = 'epochs', y = 'train_acc', data = accuracies, marker = 'x', label = 'train_acc')\nplt.scatter(x = 'epochs', y = 'val_acc', s = 5,data = accuracies, label = 'val_acc', color = 'r')\nplt.plot(accuracies.iloc[:,0:2])\nplt.legend()\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"id":"qjprIjxa4_sw","colab_type":"code","colab":{},"trusted":true,"_uuid":"8c497fda45c0acc816759e81cb9499ebd2a10247"},"cell_type":"code","source":"\nplt.scatter(x = 'epochs', y = 'train_loss', data = losses, marker = 'x', label = 'train_loss')\nplt.scatter(x = 'epochs', y = 'val_loss', data = losses, s = 5,label = 'val_loss', color = 'r')\nplt.plot(losses.iloc[:,0:2])\nplt.legend()\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"996fe21692711d4ba80ee4c23c1f04cb57eae8a8"},"cell_type":"code","source":"test = np.array(test)/255. #Need also to normalize data that we need to predict\ntest = test.reshape(-1,28,28,1)","execution_count":null,"outputs":[]},{"metadata":{"id":"5BNIJLrV5VHR","colab_type":"code","colab":{},"trusted":true,"_uuid":"971c35cdda57be01725f54e1051a7d7a817ca61c"},"cell_type":"code","source":"#Combine train and validation set for final model training, from scratch\ntotalData_X = np.concatenate((train_X, test_X))\ntotalData_y = np.concatenate((train_y, test_y))\nmyDataGen = generator.flow(totalData_X,totalData_y, batch_size = 100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"582c1fbbf05d651dd72a6be254ef4c7c22b0239e"},"cell_type":"code","source":"#Redefining model\nmodel = keras.models.Sequential()\nmodel.add(keras.layers.Conv2D(32, (3,3), input_shape = (28,28,1,), activation = 'relu', name = 'Conv2D1'))\nmodel.add(keras.layers.MaxPooling2D((2,2)))\nmodel.add(keras.layers.BatchNormalization())\nmodel.add(keras.layers.Conv2D(64, (3,3), activation = 'relu', name = 'Conv2D2'))\nmodel.add(keras.layers.MaxPooling2D(2,2))\nmodel.add(keras.layers.BatchNormalization())\nmodel.add(keras.layers.Conv2D(128, (2,2),activation = 'relu', name = 'Conv2D3'))\nmodel.add(keras.layers.Dropout(0.5))\nmodel.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(10, activation = 'softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b649baf8dc490b0509f5fd496317b6fc67057055"},"cell_type":"code","source":"model.compile(optimizer = keras.optimizers.Adam(.001), loss = 'sparse_categorical_crossentropy', metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"61c925634791170325ec6de85e7f28f3f3373d56"},"cell_type":"code","source":"hist  = model.fit_generator(myDataGen, steps_per_epoch = 450, epochs = 40)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d40908c23560a1f3f9254359400931e539a32c95"},"cell_type":"code","source":"preds = model.predict(test)","execution_count":null,"outputs":[]},{"metadata":{"id":"EN5Q4A8v5H4z","colab_type":"code","outputId":"deee237b-90e6-4db4-9af0-ad55bb93f14d","colab":{"base_uri":"https://localhost:8080/","height":111},"trusted":true,"_uuid":"aef568778354cc3d9482bdb526b0bf8a4d3749f9"},"cell_type":"code","source":"predict = pd.DataFrame([preds[x].argmax() for x in range(len(preds))], columns = ['Label'])\npredict.head(2)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"MPh9RoY_8lEh","colab_type":"code","colab":{},"trusted":true,"_uuid":"a1d70021e440f7f81d534fa7653b4bf75a2f1ae7"},"cell_type":"code","source":"sample = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"id":"EaJRUq0f86eI","colab_type":"code","colab":{},"trusted":true,"_uuid":"2bca9aacd7990beeb1e6dae6d17d9a829c2a129e"},"cell_type":"code","source":"predict['ImageId'] = sample['ImageId']\npredict = predict[['ImageId','Label']]","execution_count":null,"outputs":[]},{"metadata":{"id":"-QYpfWlU9D3a","colab_type":"code","outputId":"39ab5bf9-230a-4860-ea38-ada3d60b238f","colab":{"base_uri":"https://localhost:8080/","height":204},"trusted":true,"_uuid":"3e757f5e9a10b64b34df150b4d16a86691d9af2f"},"cell_type":"code","source":"predict.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"hS3t_q4p9FTH","colab_type":"code","colab":{},"trusted":true,"_uuid":"6762d53d420a6ecd551ce02625f16b0b18793aec"},"cell_type":"code","source":"predict.to_csv('predictionsMy.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"colab":{"name":"Untitled3.ipynb","version":"0.3.2","provenance":[]},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"accelerator":"GPU","language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}