{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T11:14:07.474562Z","iopub.execute_input":"2022-08-09T11:14:07.475659Z","iopub.status.idle":"2022-08-09T11:14:07.504725Z","shell.execute_reply.started":"2022-08-09T11:14:07.475530Z","shell.execute_reply":"2022-08-09T11:14:07.503749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten,Conv2D , MaxPool2D \nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.layers.experimental.preprocessing import Normalization\nfrom tensorflow.keras.optimizers import Adam\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:44:40.077891Z","iopub.execute_input":"2022-08-09T11:44:40.078391Z","iopub.status.idle":"2022-08-09T11:44:47.358305Z","shell.execute_reply.started":"2022-08-09T11:44:40.078350Z","shell.execute_reply":"2022-08-09T11:44:47.357308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Reading Data**","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv(\"../input/digit-recognizer/train.csv\")\ntest=pd.read_csv(\"../input/digit-recognizer/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:16:29.997836Z","iopub.execute_input":"2022-08-09T11:16:29.998333Z","iopub.status.idle":"2022-08-09T11:16:35.916446Z","shell.execute_reply.started":"2022-08-09T11:16:29.998294Z","shell.execute_reply":"2022-08-09T11:16:35.915088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"code","source":"train.shape , test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:16:56.846737Z","iopub.execute_input":"2022-08-09T11:16:56.847245Z","iopub.status.idle":"2022-08-09T11:16:56.862377Z","shell.execute_reply.started":"2022-08-09T11:16:56.847206Z","shell.execute_reply":"2022-08-09T11:16:56.860684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:17:31.372417Z","iopub.execute_input":"2022-08-09T11:17:31.372952Z","iopub.status.idle":"2022-08-09T11:17:31.400313Z","shell.execute_reply.started":"2022-08-09T11:17:31.372900Z","shell.execute_reply":"2022-08-09T11:17:31.399070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing = train.isna().sum()>0\nmissing.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:19:18.636320Z","iopub.execute_input":"2022-08-09T11:19:18.636772Z","iopub.status.idle":"2022-08-09T11:19:18.697935Z","shell.execute_reply.started":"2022-08-09T11:19:18.636729Z","shell.execute_reply":"2022-08-09T11:19:18.696736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing1 = test.isna().sum()>0\nmissing1.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:19:34.175229Z","iopub.execute_input":"2022-08-09T11:19:34.175686Z","iopub.status.idle":"2022-08-09T11:19:34.215923Z","shell.execute_reply.started":"2022-08-09T11:19:34.175650Z","shell.execute_reply":"2022-08-09T11:19:34.214953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = train.drop(\"label\", axis=1)\ny_train = train[\"label\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:35:44.002733Z","iopub.execute_input":"2022-08-09T11:35:44.003219Z","iopub.status.idle":"2022-08-09T11:35:44.188372Z","shell.execute_reply.started":"2022-08-09T11:35:44.003180Z","shell.execute_reply":"2022-08-09T11:35:44.187037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:35:44.279701Z","iopub.execute_input":"2022-08-09T11:35:44.280224Z","iopub.status.idle":"2022-08-09T11:35:44.288778Z","shell.execute_reply.started":"2022-08-09T11:35:44.280179Z","shell.execute_reply":"2022-08-09T11:35:44.287511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Preparation**","metadata":{}},{"cell_type":"code","source":"x_train = x_train.values.reshape(-1,28,28,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:35:44.513739Z","iopub.execute_input":"2022-08-09T11:35:44.514481Z","iopub.status.idle":"2022-08-09T11:35:44.521633Z","shell.execute_reply.started":"2022-08-09T11:35:44.514425Z","shell.execute_reply":"2022-08-09T11:35:44.520283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(9):\n    plt.subplot(330+1+i)\n    plt.imshow(x_train[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:35:53.004605Z","iopub.execute_input":"2022-08-09T11:35:53.005146Z","iopub.status.idle":"2022-08-09T11:35:53.777818Z","shell.execute_reply.started":"2022-08-09T11:35:53.005100Z","shell.execute_reply":"2022-08-09T11:35:53.776505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = test.values.reshape(-1,28,28,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:42:35.621583Z","iopub.execute_input":"2022-08-09T11:42:35.621968Z","iopub.status.idle":"2022-08-09T11:42:35.627649Z","shell.execute_reply.started":"2022-08-09T11:42:35.621936Z","shell.execute_reply":"2022-08-09T11:42:35.626257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:43:04.155047Z","iopub.execute_input":"2022-08-09T11:43:04.155554Z","iopub.status.idle":"2022-08-09T11:43:04.163962Z","shell.execute_reply.started":"2022-08-09T11:43:04.155515Z","shell.execute_reply":"2022-08-09T11:43:04.162911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train/255\nx_test = x_test/255\ny_train = to_categorical(y_train, num_classes = 10)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:45:21.191710Z","iopub.execute_input":"2022-08-09T11:45:21.192701Z","iopub.status.idle":"2022-08-09T11:45:21.415674Z","shell.execute_reply.started":"2022-08-09T11:45:21.192649Z","shell.execute_reply":"2022-08-09T11:45:21.414505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(x_train, y_train, test_size = 0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:46:12.011556Z","iopub.execute_input":"2022-08-09T11:46:12.012168Z","iopub.status.idle":"2022-08-09T11:46:12.093828Z","shell.execute_reply.started":"2022-08-09T11:46:12.012116Z","shell.execute_reply":"2022-08-09T11:46:12.092270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model evaluation and accuracy**","metadata":{}},{"cell_type":"code","source":"model = tf.keras.Sequential([\n                            tf.keras.layers.Conv2D(32, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same', input_shape=(28,28,1)),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Conv2D(32, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same'),\n                            tf.keras.layers.MaxPooling2D(2,2),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Dropout(0.2),\n                            tf.keras.layers.Conv2D(64, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same'), \n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Conv2D(64, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same'),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.MaxPooling2D(2,2),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Dropout(0.2),\n                            tf.keras.layers.Conv2D(128, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same'),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Conv2D(128, (3, 3), activation='relu', kernel_initializer='he_uniform', padding='same'),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.MaxPooling2D(2,2),\n                            tf.keras.layers.Dropout(0.2), \n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Flatten(),\n                            tf.keras.layers.Dense(64, activation = 'relu', kernel_initializer='he_uniform'),\n                            tf.keras.layers.BatchNormalization(),\n                            tf.keras.layers.Dropout(0.2),\n                            tf.keras.layers.Dense(10, activation = 'softmax'),])\n                           \n\nmodel.compile(optimizer ='adam', loss='categorical_crossentropy', metrics=['accuracy'])\nhistory = model.fit(x_train, y_train, epochs=10,validation_data=(x_val,y_val))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:48:49.555885Z","iopub.execute_input":"2022-08-09T11:48:49.556397Z","iopub.status.idle":"2022-08-09T11:58:02.592044Z","shell.execute_reply.started":"2022-08-09T11:48:49.556358Z","shell.execute_reply":"2022-08-09T11:58:02.590336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'], label='accuracy')\nplt.plot(history.history['val_accuracy'], label = 'val_accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.ylim([0.5, 1])\nplt.legend(loc='lower right')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:59:27.335315Z","iopub.execute_input":"2022-08-09T11:59:27.336047Z","iopub.status.idle":"2022-08-09T11:59:27.526508Z","shell.execute_reply.started":"2022-08-09T11:59:27.335999Z","shell.execute_reply":"2022-08-09T11:59:27.525138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction and submission**","metadata":{}},{"cell_type":"code","source":"results = model.predict(x_test)\nresults = np.argmax(results,axis = 1)\nresults = pd.Series(results,name=\"Label\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:00:46.872801Z","iopub.execute_input":"2022-08-09T12:00:46.873458Z","iopub.status.idle":"2022-08-09T12:01:04.758795Z","shell.execute_reply.started":"2022-08-09T12:00:46.873416Z","shell.execute_reply":"2022-08-09T12:01:04.757277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([pd.Series(range(1,28001),name = \"ImageId\"),results],axis = 1)\nsubmission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:01:08.916209Z","iopub.execute_input":"2022-08-09T12:01:08.916716Z","iopub.status.idle":"2022-08-09T12:01:08.965864Z","shell.execute_reply.started":"2022-08-09T12:01:08.916677Z","shell.execute_reply":"2022-08-09T12:01:08.964549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}