{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Digit Recognization with Keras\n## Packages\nThe first step is to import the packages.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction import DictVectorizer\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers.core import Dense, Dropout, Activation\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten\nfrom tensorflow.keras.optimizers import SGD, Adam\nfrom keras.utils import np_utils\nfrom keras.datasets import mnist","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T08:58:06.662692Z","iopub.execute_input":"2022-08-11T08:58:06.663911Z","iopub.status.idle":"2022-08-11T08:58:21.262517Z","shell.execute_reply.started":"2022-08-11T08:58:06.663749Z","shell.execute_reply":"2022-08-11T08:58:21.260650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read the Datasets\nRead the datasets of training deep learning models and predicting the test data respectively.","metadata":{}},{"cell_type":"code","source":"#load data\ndf_train = pd.read_csv('../input/digit-recognizer/train.csv')\ndf_test = pd.read_csv('../input/digit-recognizer/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:21.265716Z","iopub.execute_input":"2022-08-11T08:58:21.268266Z","iopub.status.idle":"2022-08-11T08:58:27.500476Z","shell.execute_reply.started":"2022-08-11T08:58:21.268210Z","shell.execute_reply":"2022-08-11T08:58:27.499290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, see the overview of the training data to understanding how to process the dataset.","metadata":{}},{"cell_type":"code","source":"print(type(df_train), '\\t', type(df_test))\nprint(df_train.shape, '\\t', df_test.shape)\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:27.502395Z","iopub.execute_input":"2022-08-11T08:58:27.503269Z","iopub.status.idle":"2022-08-11T08:58:27.536815Z","shell.execute_reply.started":"2022-08-11T08:58:27.503220Z","shell.execute_reply":"2022-08-11T08:58:27.535580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Processing for Training\nFirst, take apart the training data as x(features) and y(label).","metadata":{}},{"cell_type":"code","source":"x=df_train.drop(['label'], axis='columns')\ny=df_train['label']\n#df_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:27.539890Z","iopub.execute_input":"2022-08-11T08:58:27.541538Z","iopub.status.idle":"2022-08-11T08:58:27.666356Z","shell.execute_reply.started":"2022-08-11T08:58:27.541471Z","shell.execute_reply":"2022-08-11T08:58:27.664877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualize the label distribution as the bar chart below. As we can see, the labels are evenly distributed.","metadata":{}},{"cell_type":"code","source":"#data visualization: number counts\nplt.figure(figsize=(12,6))\nax=sns.countplot(x='label', data=df_train, palette='flare')\nplt.xlabel('label', fontsize=15)\nplt.ylabel('counts', fontsize=15)\n#sns.countplot(x=y)\nplt.title('label counts', fontsize=15)\nfor p in ax.patches:\n        ax.annotate('{:d}'.format(p.get_height()), (p.get_x()+0.13, p.get_height()+50), fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:27.668750Z","iopub.execute_input":"2022-08-11T08:58:27.670614Z","iopub.status.idle":"2022-08-11T08:58:27.999815Z","shell.execute_reply.started":"2022-08-11T08:58:27.670559Z","shell.execute_reply":"2022-08-11T08:58:27.998409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check missing value\nnp.where(np.isnan(df_train)) #no missing value","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:28.001889Z","iopub.execute_input":"2022-08-11T08:58:28.002384Z","iopub.status.idle":"2022-08-11T08:58:28.144003Z","shell.execute_reply.started":"2022-08-11T08:58:28.002321Z","shell.execute_reply":"2022-08-11T08:58:28.142694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use the pixel data to know what the original image looks like.","metadata":{}},{"cell_type":"code","source":"# plot sample\nplt.figure(figsize=(8,6))\nimg = x.iloc[42000-1].to_numpy()\nimg = img.reshape((28,28))\nplt.imshow(img,cmap='gray')\nplt.title(df_train.iloc[42000-1,0])\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:28.145845Z","iopub.execute_input":"2022-08-11T08:58:28.147157Z","iopub.status.idle":"2022-08-11T08:58:28.387573Z","shell.execute_reply.started":"2022-08-11T08:58:28.147100Z","shell.execute_reply":"2022-08-11T08:58:28.386108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, transfer the dataframe to numpy array and simply standardize the data for training model. Because the labels are integers, we should transfer the data type of labels to categorical data. The final step before training model is to split the data x1 and y1 for training model and testing the predicted ability of the model.","metadata":{}},{"cell_type":"code","source":"#transfer dataframe to numpy array(train data)\ndvec=DictVectorizer(sparse=False)\nx1=dvec.fit_transform(x.to_dict(orient='records'))\nx1=x1/255.0\nx1=x1.astype('float32')\n#print(x1)\n#print(x1.shape)\n#print(type(x1))\n\n#transfer y's type to categorical\ny1 = np_utils.to_categorical(y, num_classes=10)\n#print(y1)\n\nx_train, x_test, y_train, y_test=train_test_split(x1, y1, random_state=1, train_size=0.8)\nprint(x_train.shape, x_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:58:28.389748Z","iopub.execute_input":"2022-08-11T08:58:28.390267Z","iopub.status.idle":"2022-08-11T08:59:42.094599Z","shell.execute_reply.started":"2022-08-11T08:58:28.390219Z","shell.execute_reply":"2022-08-11T08:59:42.093194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training\n### Fully Connected Neural Network\nAfter the data processing, we can use the data to train the model. The first model is the basic model of deep learning, fully connected neural network. This method can be finished easily by Keras, the only thing needs to think is choosing the hyperparameters. Therefore, finish the steps squentially and use the testing data split from training data to test the predicted ability of the model. The accuracy reaches about 97%.","metadata":{}},{"cell_type":"code","source":"#DNN model\nmodel=Sequential()\nmodel.add(Dense(600, activation='relu', input_dim=28*28))\nmodel.add(Dense(10, activation='softmax'))\nmodel.summary()\n\nmodel.compile(loss='categorical_crossentropy', optimizer=SGD(learning_rate=0.1), metrics=['accuracy'])\n\nmodel.fit(x_train, y_train, batch_size=100, epochs=20)\n\nresult=model.evaluate(x_test, y_test)\nprint('prediction acc:', result[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:59:42.096380Z","iopub.execute_input":"2022-08-11T08:59:42.096842Z","iopub.status.idle":"2022-08-11T09:01:05.867212Z","shell.execute_reply.started":"2022-08-11T08:59:42.096802Z","shell.execute_reply":"2022-08-11T09:01:05.865668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prob=model.predict(x_test)\nprint(y_prob.shape)\n\ny_predict=np.argmax(y_prob, axis=1)\nprint(y_predict.shape)\n\nprint(y_predict[0:10], '\\n', np.argmax(y_test, axis=1)[0:10], '\\n', y_test[0:10])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:01:05.871438Z","iopub.execute_input":"2022-08-11T09:01:05.872958Z","iopub.status.idle":"2022-08-11T09:01:06.657008Z","shell.execute_reply.started":"2022-08-11T09:01:05.872908Z","shell.execute_reply":"2022-08-11T09:01:06.655462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convolutional Neural Network\nAfter the basic model, now, try to use the most common method for image recognization, convolutional neural network. The steps are similar to fully connected neural network. So finish the coding step by step and we can get the accuracy from the CNN model, it reaches 98% which is slightly higher than fully connected neural network.","metadata":{}},{"cell_type":"code","source":"#CNN model\nmodel2=Sequential()\nmodel2.add(Conv2D(filters=25, kernel_size=(3,3), padding='Same', activation='relu', input_shape=(28,28,1)))\nmodel2.add(MaxPooling2D(pool_size=(2,2)))\nmodel2.add(Dropout(0.25))\n\nmodel2.add(Flatten())\nmodel2.add(Dense(300, activation='relu'))\nmodel2.add(Dense(10, activation='softmax'))\n\nmodel2.summary()\n\nmodel2.compile(loss='categorical_crossentropy', optimizer=SGD(learning_rate=0.1), metrics=['accuracy'])\n\nmodel2.fit(x_train.reshape(-1,28,28,1), y_train, batch_size=100, epochs=20)\n\nmodel2.evaluate(x_test.reshape(-1,28,28,1), y_test)\n#print('prediction acc:', result2[1])\n#x_train.reshape(-1,28,28,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:01:06.658704Z","iopub.execute_input":"2022-08-11T09:01:06.659517Z","iopub.status.idle":"2022-08-11T09:04:23.911866Z","shell.execute_reply.started":"2022-08-11T09:01:06.659465Z","shell.execute_reply":"2022-08-11T09:04:23.910012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Processing for Testing\nThis step is as the same as the process for training. Transform the dataframe type to numpy array, and simply standardize the data.","metadata":{}},{"cell_type":"code","source":"#transfer dataframe to numpy array(test data)\ndvec=DictVectorizer(sparse=False)\nx_testdata=dvec.fit_transform(df_test.to_dict(orient='records'))\nx_testdata=x_testdata/255.0\nx_testdata=x_testdata.astype('float32')\nprint(np.max(x_testdata[0]))\nprint(x_testdata.shape)\nprint(type(x_testdata))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:04:23.914468Z","iopub.execute_input":"2022-08-11T09:04:23.914960Z","iopub.status.idle":"2022-08-11T09:05:13.962079Z","shell.execute_reply.started":"2022-08-11T09:04:23.914915Z","shell.execute_reply":"2022-08-11T09:05:13.960271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction\nAccording the steps in the following, we can get the results of prediction from the two models trained previously. Then, save the output as .csv files and submit the output on Kaggle.","metadata":{}},{"cell_type":"code","source":"#prediction of test data (model: DNN)\ny_testdata_predict=model.predict(x_testdata)\ny_testdata_predict=np.argmax(y_testdata_predict, axis=1)\n\nprint(y_testdata_predict.shape)\n\nprint(df_test.shape[0])\n\noutput1=pd.DataFrame({\n    'ImageId':np.arange(1, df_test.shape[0]+1),\n    'Label':y_testdata_predict\n})\n\nprint(output1.head(5))\n#output1.to_csv(\"submission_DNN.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:05:13.964310Z","iopub.execute_input":"2022-08-11T09:05:13.964736Z","iopub.status.idle":"2022-08-11T09:05:16.204031Z","shell.execute_reply.started":"2022-08-11T09:05:13.964696Z","shell.execute_reply":"2022-08-11T09:05:16.202459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction of test data(model2: CNN)\ny_testdata_predict2=model2.predict(x_testdata.reshape(-1,28,28,1))\ny_testdata_predict2=np.argmax(y_testdata_predict2, axis=1)\n\nprint(y_testdata_predict2.shape)\n\nprint(df_test.shape[0])\n\noutput2=pd.DataFrame({\n    'ImageId':np.arange(1, df_test.shape[0]+1),\n    'Label':y_testdata_predict2\n})\n\nprint(output2.head(5))\noutput2.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:05:16.206313Z","iopub.execute_input":"2022-08-11T09:05:16.206782Z","iopub.status.idle":"2022-08-11T09:05:20.579115Z","shell.execute_reply.started":"2022-08-11T09:05:16.206741Z","shell.execute_reply":"2022-08-11T09:05:20.577714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion\nThis is a simple practice of deep learning. Throughout the process of each step, we can understand the concepts completely. Although the practice seems easy to learn and deep learning seems much more complicated, it is still an important course in this field. ","metadata":{}}]}