{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Digit Recognizer<center>","metadata":{}},{"cell_type":"markdown","source":"The objective of this work is to present a solution for Kaggle's Digit Recognizer competition using Artificial Neural Networks.","metadata":{}},{"cell_type":"markdown","source":"More information about the competition at https://www.kaggle.com/competitions/digit-recognizer/overview","metadata":{}},{"cell_type":"code","source":"# Some basics importations\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom keras.layers import Dense , BatchNormalization\nfrom keras.models import Sequential\nfrom keras.callbacks import ModelCheckpoint , EarlyStopping\nfrom IPython.display import Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the Data","metadata":{}},{"cell_type":"code","source":"# Reading the training file\n\ntrain = pd.read_csv('../input/digit-recognizer/train.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dimension of training dataset\n\ntrain.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Observing some training data\n\ntrain.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for missing values\n\ntrain.isnull().sum()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Total missing values\n\nsum(train.isnull().sum())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading the test file\n\ntest = pd.read_csv('../input/digit-recognizer/test.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dimension of test data\n\ntest.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are 0 missing values in the test set !\n\nsum(test.isnull().sum())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = np.array(train['label'])\nX_train = train.drop('label' , axis = 1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One-Hot-Encoding\n\nlista = [0]*10\ny_train_encoding = []\nfor i in y_train:\n    lista[i] = 1\n    y_train_encoding.append(lista)\n    lista = 10*[0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_encoding = np.array(y_train_encoding)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_encoding","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center>Artificial Neural Network<center>","metadata":{}},{"cell_type":"code","source":"#  Building the network\n\nmodel = Sequential()\nmodel.add(Dense(350 , activation = 'relu' , input_shape = (784 , )))\nmodel.add(BatchNormalization())\nmodel.add(Dense(200 , activation = 'relu' ))\nmodel.add(BatchNormalization())\nmodel.add(Dense(100 , activation = 'relu' ))\nmodel.add(BatchNormalization())\nmodel.add(Dense(50 , activation = 'relu' ))\nmodel.add(BatchNormalization())\nmodel.add(Dense(25 , activation = 'relu' ))\nmodel.add(BatchNormalization())\nmodel.add(Dense(10 , activation = 'relu' ))\nmodel.add(BatchNormalization())\nmodel.add(Dense(10 , activation = 'softmax'))\nmodel.compile(optimizer = 'adam' , loss = 'categorical_crossentropy' , metrics = ['accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style='text-align: justify;'>The neural network built above has 7 dense layers and in all these layers with the exception of the last one, the activation function is relu. In the last layer the activation function is softmax, as our task is a multi-category classification task and this function will give the probability of each class. Finally, the chosen optimizer is adam.","metadata":{}},{"cell_type":"code","source":"# Creating a checkpoint to save the model weights that lead to the highest validation accuracy\n\ncheckpoint = ModelCheckpoint('weights.hdf5' , monitor = 'val_accuracy' , save_best_only = True )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Forcing the model to stop when 5 epochs have passed without increasing the validation accuracy value\n\nearly_stopping = EarlyStopping(monitor = 'val_accuracy' , patience = 5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the neural network\n\nhistory = model.fit(X_train , y_train_encoding, epochs = 40 , validation_split = 0.2 , callbacks = [checkpoint, early_stopping])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Higher validation accuracy value\n\nmax(history.history['val_accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of epochs\n\nn_epochs = len(history.history['val_accuracy'])\n\nn_epochs","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the weights that generate the highest validation accuracy\n\nmodel.load_weights('weights.hdf5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(list(range(1 , n_epochs + 1, 1)) , history.history['accuracy'] , label = 'accuracy')\nplt.plot(list(range(1 , n_epochs + 1, 1)) , history.history['val_accuracy'] , label = 'val_accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center>Submission<center>","metadata":{}},{"cell_type":"code","source":"X_test = np.array(test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = np.argmax(model.predict(X_test), axis=-1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Label'] = predictions\ntest['ImageId'] = list(range(1, 28001 , 1))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['ImageId' , 'Label']].to_csv('submission.csv' , index = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This submission gave us a Kaggle score of  0.97628, so our validation seems to make sense.","metadata":{}}]}