{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Loading the packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow.keras.layers as tfl\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:37.414554Z","iopub.execute_input":"2022-07-08T22:44:37.415246Z","iopub.status.idle":"2022-07-08T22:44:50.153563Z","shell.execute_reply.started":"2022-07-08T22:44:37.415076Z","shell.execute_reply":"2022-07-08T22:44:50.151983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.1. Loading the datasets","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/digit-recognizer/train.csv')\ntest_data = pd.read_csv('../input/digit-recognizer/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:50.156134Z","iopub.execute_input":"2022-07-08T22:44:50.15689Z","iopub.status.idle":"2022-07-08T22:44:56.69033Z","shell.execute_reply.started":"2022-07-08T22:44:50.156831Z","shell.execute_reply":"2022-07-08T22:44:56.689099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inspecting the dataset shape\nprint(train_data.shape)\nprint(test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.691743Z","iopub.execute_input":"2022-07-08T22:44:56.692091Z","iopub.status.idle":"2022-07-08T22:44:56.698841Z","shell.execute_reply.started":"2022-07-08T22:44:56.69206Z","shell.execute_reply":"2022-07-08T22:44:56.697253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Preprocessing the dataset","metadata":{}},{"cell_type":"markdown","source":"### 2.1. Spliting features and target variables from train dataset","metadata":{}},{"cell_type":"code","source":"train_labels = train_data['label']\ntrain_data = train_data.drop('label', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.7024Z","iopub.execute_input":"2022-07-08T22:44:56.703193Z","iopub.status.idle":"2022-07-08T22:44:56.853876Z","shell.execute_reply.started":"2022-07-08T22:44:56.703142Z","shell.execute_reply":"2022-07-08T22:44:56.852658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2. Encoding the target values to numerical using OneHotEncoder","metadata":{}},{"cell_type":"code","source":"encoder = OneHotEncoder()\ny = encoder.fit_transform(train_labels.values.reshape(-1, 1)).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.855624Z","iopub.execute_input":"2022-07-08T22:44:56.856102Z","iopub.status.idle":"2022-07-08T22:44:56.869634Z","shell.execute_reply.started":"2022-07-08T22:44:56.856055Z","shell.execute_reply":"2022-07-08T22:44:56.868719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.871192Z","iopub.execute_input":"2022-07-08T22:44:56.872409Z","iopub.status.idle":"2022-07-08T22:44:56.899163Z","shell.execute_reply.started":"2022-07-08T22:44:56.872358Z","shell.execute_reply":"2022-07-08T22:44:56.898028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.3. Reshaping the train data features of size 784 into (28, 28) for using CNN","metadata":{}},{"cell_type":"markdown","source":"The images are reshaped to 2D grayscale to better suit the model which uses CNN. The CNN model follows LeNet approach and therefore will again be reshaped to (32, 32) later in this notebook.","metadata":{}},{"cell_type":"code","source":"train_data = train_data.to_numpy().reshape(train_data.shape[0], 28, 28)\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.900814Z","iopub.execute_input":"2022-07-08T22:44:56.901191Z","iopub.status.idle":"2022-07-08T22:44:56.908087Z","shell.execute_reply.started":"2022-07-08T22:44:56.901157Z","shell.execute_reply":"2022-07-08T22:44:56.907216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing the first 9 images in the train dataset\nplt.figure(figsize = (10, 10))\nplt.subplots_adjust(wspace = 0.5)\nfor i in range(1, 10):\n    plt.subplot(3, 3, i)\n    plt.imshow(train_data[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:56.909368Z","iopub.execute_input":"2022-07-08T22:44:56.910143Z","iopub.status.idle":"2022-07-08T22:44:57.834364Z","shell.execute_reply.started":"2022-07-08T22:44:56.910107Z","shell.execute_reply":"2022-07-08T22:44:57.832632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Defining the model for classification","metadata":{}},{"cell_type":"markdown","source":"As mentioned in section 2.3., the images are reshaped to (32, 32) which is the input shape used for LeNet approach v1. It is followed by data augmentation using RandomZoom. Random zoom is chosen as a data augmentation method after inspecting the misclassifed records through error analysis mentioned in section 4. The model then contains the following steps,\n* 2D Convolution with 5 x 5 filter and stride of (1, 1)\n* 2D Average Pooling with strides of (2, 2)\n* 2D Convolution with filter size 5 x 5 and stride (1, 1)\n* 2D Average Pooling with strides 2 on each direction\n* Fully Connected layer (Dense) with 120 hidden units\n* Fully Connected layer (Dense) with 84 hidden units\n* Softmax layer with 10 output units for each digit","metadata":{}},{"cell_type":"code","source":"def digit_classifier():\n    \n    # input layer\n    input_data = tf.keras.Input(shape = (28, 28, 1))\n    \n    # preprocessing by resizing the image to (32, 32) and using random zoom for data augmentation\n    X = tfl.Resizing(32, 32)(input_data)\n    X = tfl.RandomZoom(0.2)(X)\n    \n    # layer 1\n    X = tfl.Conv2D(filters = 6, kernel_size = (5, 5), strides = (1, 1))(X)\n    X = tfl.BatchNormalization()(X)\n    X = tfl.Activation('relu')(X)\n    \n    # layer 2\n    X = tfl.AveragePooling2D(strides = 2)(X)\n    \n    # layer 3\n    X = tfl.Conv2D(16, 5, 1)(X)\n    X = tfl.BatchNormalization()(X)\n    X = tfl.Activation('relu')(X)\n    \n    # layer 4\n    X = tfl.AveragePooling2D(strides = 2)(X)\n    \n    # layer 5\n    X = tfl.Flatten()(X)\n    X = tfl.Dense(120, activation = 'relu')(X)\n    \n    # layer 6\n    X = tfl.Dense(84, activation = 'sigmoid')(X)\n    \n    # output layer\n    output = tfl.Dense(10, activation = 'softmax')(X)\n    \n    model = tf.keras.Model(inputs = input_data, outputs = output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:57.836519Z","iopub.execute_input":"2022-07-08T22:44:57.837003Z","iopub.status.idle":"2022-07-08T22:44:57.851794Z","shell.execute_reply.started":"2022-07-08T22:44:57.836963Z","shell.execute_reply":"2022-07-08T22:44:57.850134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1. Inspecting the model summary","metadata":{}},{"cell_type":"code","source":"clf = digit_classifier()\nclf.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:57.856909Z","iopub.execute_input":"2022-07-08T22:44:57.857619Z","iopub.status.idle":"2022-07-08T22:44:58.160305Z","shell.execute_reply.started":"2022-07-08T22:44:57.857487Z","shell.execute_reply":"2022-07-08T22:44:58.158053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.2. Compling the model","metadata":{}},{"cell_type":"markdown","source":"The model uses Adam optimizer for backpropagation and Binary Crossentropy for loss calculation.","metadata":{}},{"cell_type":"code","source":"clf.compile(\n    optimizer = tf.keras.optimizers.Adam(),\n    loss = tf.keras.losses.BinaryCrossentropy(),\n    metrics = ['accuracy']\n           )","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:58.161771Z","iopub.execute_input":"2022-07-08T22:44:58.162195Z","iopub.status.idle":"2022-07-08T22:44:58.186613Z","shell.execute_reply.started":"2022-07-08T22:44:58.162158Z","shell.execute_reply":"2022-07-08T22:44:58.185459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(train_data, y, epochs = 20, validation_split = 0.05)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:44:58.187889Z","iopub.execute_input":"2022-07-08T22:44:58.189409Z","iopub.status.idle":"2022-07-08T22:52:22.054631Z","shell.execute_reply.started":"2022-07-08T22:44:58.18936Z","shell.execute_reply":"2022-07-08T22:52:22.053185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.3. Evaluate the performance of the model on each epoch","metadata":{}},{"cell_type":"code","source":"history = clf.history.history\nloss_metrics = pd.DataFrame({\n    'loss' : history['loss'],\n    'val_loss' : history['val_loss']\n})\naccuracy_metrics = pd.DataFrame({\n    'accuracy' : history['accuracy'],\n    'val_accuracy' : history['val_accuracy']\n})","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:22.05664Z","iopub.execute_input":"2022-07-08T22:52:22.057081Z","iopub.status.idle":"2022-07-08T22:52:22.067018Z","shell.execute_reply.started":"2022-07-08T22:52:22.057039Z","shell.execute_reply":"2022-07-08T22:52:22.065732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.3.1. Plotting loss after each epoch","metadata":{}},{"cell_type":"code","source":"loss_metrics.plot(xlabel = 'Epochs', ylabel = 'Loss')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:22.068564Z","iopub.execute_input":"2022-07-08T22:52:22.068974Z","iopub.status.idle":"2022-07-08T22:52:22.598335Z","shell.execute_reply.started":"2022-07-08T22:52:22.068935Z","shell.execute_reply":"2022-07-08T22:52:22.596962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.3.2. Plotting accuracy after each epoch","metadata":{}},{"cell_type":"code","source":"accuracy_metrics.plot(xlabel = 'Epochs', ylabel = 'Accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:22.600337Z","iopub.execute_input":"2022-07-08T22:52:22.601217Z","iopub.status.idle":"2022-07-08T22:52:22.828085Z","shell.execute_reply.started":"2022-07-08T22:52:22.601164Z","shell.execute_reply":"2022-07-08T22:52:22.827079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Error Analysis","metadata":{}},{"cell_type":"markdown","source":"### 4.1. Predict the training dataset with the model","metadata":{}},{"cell_type":"code","source":"y_train_pred = clf.predict(train_data)\ny_train = y_train_pred.argmax(axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:22.82933Z","iopub.execute_input":"2022-07-08T22:52:22.830261Z","iopub.status.idle":"2022-07-08T22:52:27.992323Z","shell.execute_reply.started":"2022-07-08T22:52:22.830223Z","shell.execute_reply":"2022-07-08T22:52:27.99126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.2. Find the misclassified records","metadata":{}},{"cell_type":"markdown","source":"Based on the predicted results on training set, the indices of the misclassified records are identified. Then 100 records out of them are randomly chosen for error analysis.","metadata":{}},{"cell_type":"code","source":"indices = np.where((y_train != train_labels))[0]\nwrong_index = np.random.randint(len(indices), size = 100)\nmisclassified_records = indices[wrong_index]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:27.994333Z","iopub.execute_input":"2022-07-08T22:52:27.995218Z","iopub.status.idle":"2022-07-08T22:52:28.002662Z","shell.execute_reply.started":"2022-07-08T22:52:27.995168Z","shell.execute_reply":"2022-07-08T22:52:28.001771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.3. Inspect 9 randomly selected misclassified records","metadata":{}},{"cell_type":"markdown","source":"For the sake of simplicity, in this notebook just 9 records out of the 100 randomly selected misclassified records are visualized.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15, 15))\nplt.subplots_adjust(hspace = 0.5)\nfor i in range(1, 10):\n    index = misclassified_records[i]\n    plt.subplot(3, 3, i)\n    plt.title('actual : ' + str(train_labels[index]) + ' predicted : ' + str(y_train[index]))\n    plt.imshow(train_data[index])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:28.004645Z","iopub.execute_input":"2022-07-08T22:52:28.005459Z","iopub.status.idle":"2022-07-08T22:52:29.026275Z","shell.execute_reply.started":"2022-07-08T22:52:28.005413Z","shell.execute_reply":"2022-07-08T22:52:29.024712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Prediction","metadata":{}},{"cell_type":"markdown","source":"The test data are reshaped to 2D before using the model for prediction as it is the required input shape for the constructed model.","metadata":{}},{"cell_type":"code","source":"test_data = test_data.to_numpy().reshape(test_data.shape[0], 28, 28)\n    \ny_pred = clf.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:29.028565Z","iopub.execute_input":"2022-07-08T22:52:29.029461Z","iopub.status.idle":"2022-07-08T22:52:32.51478Z","shell.execute_reply.started":"2022-07-08T22:52:29.029409Z","shell.execute_reply":"2022-07-08T22:52:32.513921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:32.516384Z","iopub.execute_input":"2022-07-08T22:52:32.517054Z","iopub.status.idle":"2022-07-08T22:52:32.52575Z","shell.execute_reply.started":"2022-07-08T22:52:32.517008Z","shell.execute_reply":"2022-07-08T22:52:32.525006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'ImageId' : np.arange(1, y_pred.shape[0]+1),\n    'Label' : y_pred.argmax(axis = 1)\n})","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:32.527256Z","iopub.execute_input":"2022-07-08T22:52:32.52821Z","iopub.status.idle":"2022-07-08T22:52:32.536272Z","shell.execute_reply.started":"2022-07-08T22:52:32.528167Z","shell.execute_reply":"2022-07-08T22:52:32.53524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:32.539709Z","iopub.execute_input":"2022-07-08T22:52:32.540499Z","iopub.status.idle":"2022-07-08T22:52:32.550249Z","shell.execute_reply.started":"2022-07-08T22:52:32.540461Z","shell.execute_reply":"2022-07-08T22:52:32.549382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:52:32.551672Z","iopub.execute_input":"2022-07-08T22:52:32.552612Z","iopub.status.idle":"2022-07-08T22:52:32.606923Z","shell.execute_reply.started":"2022-07-08T22:52:32.552577Z","shell.execute_reply":"2022-07-08T22:52:32.606072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Other works","metadata":{}},{"cell_type":"markdown","source":"* [Titanic survival prediction](https://www.kaggle.com/code/tonyrubanrajraja/titanic-survival-prediction-using-nn)\n* [Spaceship titanic prediction](https://www.kaggle.com/code/tonyrubanrajraja/spaceship-titanic-data-analysis-and-prediction)\n* [Email classification](https://www.kaggle.com/code/tonyrubanrajraja/spam-email-classification)","metadata":{}}]}