{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('whitegrid')\n\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Conv2D, Dense, Dropout, BatchNormalization, MaxPooling2D, Flatten\nfrom tensorflow.keras.utils import to_categorical, plot_model\nfrom sklearn.model_selection import train_test_split\n\nfrom IPython.display import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T23:01:55.980914Z","iopub.execute_input":"2022-07-19T23:01:55.981381Z","iopub.status.idle":"2022-07-19T23:02:04.674804Z","shell.execute_reply.started":"2022-07-19T23:01:55.981283Z","shell.execute_reply":"2022-07-19T23:02:04.671966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Dataset","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/digit-recognizer/train.csv\")\ndf_test = pd.read_csv(\"../input/digit-recognizer/test.csv\")\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.676310Z","iopub.status.idle":"2022-07-19T23:02:04.677029Z","shell.execute_reply.started":"2022-07-19T23:02:04.676821Z","shell.execute_reply":"2022-07-19T23:02:04.676843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking Dataset Size","metadata":{}},{"cell_type":"code","source":"print(f\"Train data shape: {df_train.shape} \")\nprint(f\"Test data shape: {df_test.shape} \")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.678128Z","iopub.status.idle":"2022-07-19T23:02:04.678702Z","shell.execute_reply.started":"2022-07-19T23:02:04.678497Z","shell.execute_reply":"2022-07-19T23:02:04.678515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this dataset, we have 42000 train data and 28000 test data. Each data point on train data has 785 columns, where the first column contains the label of the data point and the remaining columns represent the value of each pixel. Unlike train data, the test data has no label column, so it only has 784 columns.","metadata":{}},{"cell_type":"markdown","source":"### Checking Target Distribution","metadata":{}},{"cell_type":"code","source":"labels = sorted(df_train[\"label\"].unique())\n\n# Create subplots with 1 row and 2 columns\nfig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\n# Plot target distribution using pie chart \ndf_train[\"label\"].value_counts().plot(kind=\"pie\", explode=[0.05 for x in labels], labels=labels, autopct='%1.1f%%', ax=ax[0], shadow=True)\nax[0].set_title(\"Target Distribution - Pie Chart\")\nax[0].set_ylabel('label')\n\n# Plot target distribution using bar chart\ncount = sns.countplot(x=\"label\", data=df_train, ax=ax[1])\n\n# Add value labels on bar chart \nfor bar in count.patches:\n    count.annotate(format(bar.get_height()),\n        (bar.get_x() + bar.get_width() / 2,\n       bar.get_height()), ha='center', va='center',\n       size=11, xytext=(0, 8),\n       textcoords='offset points')\nax[1].set_title(\"Target Distribution - Bar Chart\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.679739Z","iopub.status.idle":"2022-07-19T23:02:04.680280Z","shell.execute_reply.started":"2022-07-19T23:02:04.680088Z","shell.execute_reply":"2022-07-19T23:02:04.680105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like our target distribution is quite balanced. Label with the highest number of data is 1 with a total of 4684 data, while the least is 5 with a total of 3795 data.","metadata":{}},{"cell_type":"markdown","source":"### Visualizing The Data","metadata":{}},{"cell_type":"markdown","source":"Let's visualize our first 9 data.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 3, figsize=(12,12))\nax = ax.flatten()\n\nfor i in range(9):\n    digit = df_train.iloc[i, 1:].to_numpy().reshape(28,28)\n    ax[i].imshow(digit, cmap=\"gray\")\n    ax[i].axis(\"off\")\n    ax[i].set_title(f\"Digit: {df_train.iloc[i, 0]}\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.681276Z","iopub.status.idle":"2022-07-19T23:02:04.681815Z","shell.execute_reply.started":"2022-07-19T23:02:04.681631Z","shell.execute_reply":"2022-07-19T23:02:04.681650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here's more detailed image of the pixel value that represent the digit.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nsns.heatmap(df_train.iloc[0, 1:].to_numpy().reshape(28, 28), fmt=\".3g\", annot=True, cbar=False, cmap=\"gray\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.682798Z","iopub.status.idle":"2022-07-19T23:02:04.683343Z","shell.execute_reply.started":"2022-07-19T23:02:04.683145Z","shell.execute_reply":"2022-07-19T23:02:04.683173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting Dataset","metadata":{}},{"cell_type":"markdown","source":"We will split train data into 80% train set and 20% validation set. ","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(\"label\", axis=1)\ny = df_train[\"label\"]\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=0)\nX_test = df_test.copy()\n\nprint(f\"Train data shape: {X_train.shape}\")\nprint(f\"Validation data shape: {X_val.shape}\")\nprint(f\"Test data shape: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.685128Z","iopub.status.idle":"2022-07-19T23:02:04.687219Z","shell.execute_reply.started":"2022-07-19T23:02:04.686486Z","shell.execute_reply":"2022-07-19T23:02:04.686575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Grayscale Normalization","metadata":{}},{"cell_type":"markdown","source":"After splitting dataset, we will do grayscale normalization to scale the data values to lie between 0 and 1. This will make the training process faster.","metadata":{}},{"cell_type":"code","source":"X_train = X_train / 255\nX_val = X_val / 255\nX_test = X_test / 255","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.691195Z","iopub.status.idle":"2022-07-19T23:02:04.692871Z","shell.execute_reply.started":"2022-07-19T23:02:04.692624Z","shell.execute_reply":"2022-07-19T23:02:04.692659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Reshaping","metadata":{}},{"cell_type":"markdown","source":"We will also reshape our data into 3 dimensions.","metadata":{}},{"cell_type":"code","source":"# Reshape data into 3 dimensions (28x28x1)\nX_train = X_train.values.reshape(-1, 28, 28, 1) # -1 means all row\nX_val = X_val.values.reshape(-1, 28, 28, 1)\nX_test = X_test.values.reshape(-1, 28, 28, 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.694082Z","iopub.status.idle":"2022-07-19T23:02:04.694694Z","shell.execute_reply.started":"2022-07-19T23:02:04.694457Z","shell.execute_reply":"2022-07-19T23:02:04.694481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### One-Hot Encoding","metadata":{}},{"cell_type":"markdown","source":"To be able to work with deep learning algorithm, we need to encode our target/class variable first using one-hot encoding. In this process, we will converts a class value to a vector with a length equal to the number of unique class variable. The vector is all zeroes except in the respective category index.","metadata":{}},{"cell_type":"code","source":"num_class = y_train.nunique()\n\ny_train = to_categorical(y_train, num_classes=num_class)\ny_val = to_categorical(y_val, num_classes=num_class)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.695818Z","iopub.status.idle":"2022-07-19T23:02:04.696408Z","shell.execute_reply.started":"2022-07-19T23:02:04.696188Z","shell.execute_reply":"2022-07-19T23:02:04.696218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Building","metadata":{}},{"cell_type":"markdown","source":"In this case, we will use Convolutional Neural Network (CNN) algorithm to classify each digit in this dataset. Let's build our CNN model first.","metadata":{}},{"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(64, kernel_size=(3, 3), padding=\"same\", input_shape=(28, 28, 1), activation=\"relu\"))\nmodel.add(BatchNormalization())\n\nmodel.add(Conv2D(64, kernel_size=(3, 3), padding=\"same\", activation=\"relu\"))\nmodel.add(BatchNormalization())\n\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(128, kernel_size=(3, 3), padding=\"same\", activation=\"relu\"))\nmodel.add(BatchNormalization())\n\nmodel.add(Conv2D(128, kernel_size=(3, 3), padding=\"same\", activation=\"relu\"))\nmodel.add(BatchNormalization())\n\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(256, kernel_size=(3, 3), padding=\"same\", activation=\"relu\"))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation=\"relu\"))\nmodel.add(Dense(10, activation=\"softmax\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.697516Z","iopub.status.idle":"2022-07-19T23:02:04.698100Z","shell.execute_reply.started":"2022-07-19T23:02:04.697889Z","shell.execute_reply":"2022-07-19T23:02:04.697916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is how our CNN architecture looks like.","metadata":{}},{"cell_type":"code","source":"# Display CNN Architecture\nplot_model(model, to_file=\"model.png\", show_shapes=True, show_layer_names=True)\nImage(\"model.png\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.699175Z","iopub.status.idle":"2022-07-19T23:02:04.699742Z","shell.execute_reply.started":"2022-07-19T23:02:04.699521Z","shell.execute_reply":"2022-07-19T23:02:04.699560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Compile Model","metadata":{}},{"cell_type":"markdown","source":"After creating CNN model, the next step is to compile and train the model.","metadata":{}},{"cell_type":"code","source":"# compile CNN model\nmodel.compile(optimizer=\"adam\", loss=\"categorical_crossentropy\", metrics=\"accuracy\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.700794Z","iopub.status.idle":"2022-07-19T23:02:04.701361Z","shell.execute_reply.started":"2022-07-19T23:02:04.701144Z","shell.execute_reply":"2022-07-19T23:02:04.701181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define each parameter\nepoch = 20\nbatch_size = 64\ntrain_steps = int(X_train.shape[0] // batch_size)\nval_steps = int(X_val.shape[0] // batch_size)\n\n# train model\ntrain = model.fit(\n    X_train, \n    y_train, \n    epochs=epoch, \n    steps_per_epoch=train_steps,\n    validation_data=(X_val, y_val),\n    validation_steps=val_steps,\n    batch_size=batch_size,\n    verbose=2\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.702396Z","iopub.status.idle":"2022-07-19T23:02:04.702965Z","shell.execute_reply.started":"2022-07-19T23:02:04.702753Z","shell.execute_reply":"2022-07-19T23:02:04.702778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's plot our model loss and accuracy to get better insight.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 1, figsize=(12, 8))\n\n# plot training and validation loss\nax[0].plot(train.history[\"loss\"], color=\"blue\", label=\"Training Loss\")\nax[0].plot(train.history[\"val_loss\"], color=\"red\", label=\"Validation Loss\")\nax[0].set_title(\"Training and Validation Loss\")\nax[0].legend(loc=\"best\")\n\n# plot training and validation accuracy\nax[1].plot(train.history[\"accuracy\"], color=\"blue\", label=\"Training Accuracy\")\nax[1].plot(train.history[\"val_accuracy\"], color=\"red\", label=\"Validation Accuracy\")\nax[1].set_title(\"Training and Validation Accuracy\")\nax[1].legend(loc=\"best\")\n\nplt.subplots_adjust(hspace=0.3) # adjust horizontal space\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.704023Z","iopub.status.idle":"2022-07-19T23:02:04.704617Z","shell.execute_reply.started":"2022-07-19T23:02:04.704390Z","shell.execute_reply":"2022-07-19T23:02:04.704417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good. Our loss tends to decrease and the accuracy tends to increase as the epoch increases. The model also has a fairly good generalization and does not overfit the training data.","metadata":{}},{"cell_type":"markdown","source":"### Making Submission","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(X_test)\ny_pred = np.argmax(y_pred, axis=1)\n\nsubmission = pd.DataFrame({\n    'ImageId': [i for i in range(1, 28001)], \n    'Label': y_pred\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T23:02:04.707166Z","iopub.status.idle":"2022-07-19T23:02:04.707810Z","shell.execute_reply.started":"2022-07-19T23:02:04.707586Z","shell.execute_reply":"2022-07-19T23:02:04.707616Z"},"trusted":true},"execution_count":null,"outputs":[]}]}