{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport keras\nfrom keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:32:23.604984Z","iopub.execute_input":"2022-08-13T11:32:23.605626Z","iopub.status.idle":"2022-08-13T11:32:31.197931Z","shell.execute_reply.started":"2022-08-13T11:32:23.605518Z","shell.execute_reply":"2022-08-13T11:32:31.196874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing data and creating required arrays**","metadata":{}},{"cell_type":"code","source":"train_path = pd.read_csv(\"../input/digit-recognizer/train.csv\")\ntest_path = pd.read_csv(\"../input/digit-recognizer/test.csv\")\n\nprint (train_path.shape)\nprint (test_path.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:32:42.107925Z","iopub.execute_input":"2022-08-13T11:32:42.108533Z","iopub.status.idle":"2022-08-13T11:32:48.224614Z","shell.execute_reply.started":"2022-08-13T11:32:42.108456Z","shell.execute_reply":"2022-08-13T11:32:48.222215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = np.array(train_path.iloc[:,1:])\ntrain_x = []\ntrain_y = train_path.iloc[:,0:1]\n\ntest_data = np.array(test_path)\ntest_x = []\n\n#Reshaping the training data\nfor i in range(42000):\n    train_x.append(train_data[i].reshape(28,28,1))\n\n#Reshaping the test data\nfor i in range(28000):\n    test_x.append(test_data[i].reshape(28,28,1))\n\n#Converting list to array\ntrain_x = np.array(train_x)\ntrain_y = np.array(train_y)\ntest_x = np.array(test_x)/255\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:33:11.680851Z","iopub.execute_input":"2022-08-13T11:33:11.681158Z","iopub.status.idle":"2022-08-13T11:33:12.598010Z","shell.execute_reply.started":"2022-08-13T11:33:11.681127Z","shell.execute_reply":"2022-08-13T11:33:12.596925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visulaising the training data\n\nplt.figure(figsize=(20,10))\nfor i in range(10):\n    plt.subplot(1,10,i+1)\n    plt.axis('Off')\n    plt.imshow(train_x[i],cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:33:16.824737Z","iopub.execute_input":"2022-08-13T11:33:16.825065Z","iopub.status.idle":"2022-08-13T11:33:17.323343Z","shell.execute_reply.started":"2022-08-13T11:33:16.825033Z","shell.execute_reply":"2022-08-13T11:33:17.321945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modifying data using keras ImageDataGenerator**","metadata":{}},{"cell_type":"code","source":"#Data Augmentation\nBatch_size = 32\n\ntrain_datagen = ImageDataGenerator(rescale = 1./ 255, \n                             rotation_range = 30,  \n                             height_shift_range = 0.3,\n                             validation_split=0.3  #30% of the training data will be used for validation\n                            )\n\n                            \ntrain_datagen.fit(train_x)\n\n\n\n\naugmented_training_data = train_datagen.flow(train_x,train_y, \n                                             batch_size = Batch_size, \n                                             subset = \"training\",\n                                             seed = 2021,\n                                             shuffle = True)\n\nvalidation_data = train_datagen.flow(train_x,train_y, \n                                     batch_size = Batch_size,\n                                     subset = \"validation\",\n                                     seed = 2021,\n                                     shuffle = True)\n\nvisual_data = train_datagen.flow(train_x)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:33:36.055088Z","iopub.execute_input":"2022-08-13T11:33:36.056167Z","iopub.status.idle":"2022-08-13T11:33:36.953085Z","shell.execute_reply.started":"2022-08-13T11:33:36.056107Z","shell.execute_reply":"2022-08-13T11:33:36.951861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualising Augmented Data\nplt.figure(figsize = (10,10))\nfor i in range(5):\n    plt.subplot(1,5, i + 1)\n    plt.axis('Off')\n    plt.imshow(visual_data[i][4],cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:33:39.087071Z","iopub.execute_input":"2022-08-13T11:33:39.087411Z","iopub.status.idle":"2022-08-13T11:33:39.392474Z","shell.execute_reply.started":"2022-08-13T11:33:39.087379Z","shell.execute_reply":"2022-08-13T11:33:39.391054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building the CNN Model**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Dropout, Flatten, Input, BatchNormalization\n\ncnn_model = Sequential([\n    Input(shape = (28, 28, 1)),\n    Conv2D(32,(3,3),padding = 'same',activation = 'relu'),\n    MaxPooling2D(2,2),\n    Conv2D(64,(3,3),padding = 'same',activation = 'relu'),\n    BatchNormalization(),\n    MaxPooling2D(2,2),\n    Conv2D(128,(3,3),padding = 'same',activation = 'relu'),\n    BatchNormalization(),\n    MaxPooling2D(2,2),\n    Flatten(),\n    Dense(126, activation = 'relu'),\n    Dense(126, activation = 'relu'),\n    \n    Dense(10, activation = 'softmax')\n])\n\ncnn_model.compile(loss = 'sparse_categorical_crossentropy',\n                 optimizer = tf.keras.optimizers.Adam(0.001),\n                 metrics=['sparse_categorical_accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:34:00.495284Z","iopub.execute_input":"2022-08-13T11:34:00.495609Z","iopub.status.idle":"2022-08-13T11:34:00.738344Z","shell.execute_reply.started":"2022-08-13T11:34:00.495573Z","shell.execute_reply":"2022-08-13T11:34:00.737337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks=[\n    tf.keras.callbacks.EarlyStopping(patience=10,monitor='val_loss'),\n    tf.keras.callbacks.ReduceLROnPlateau( monitor=\"val_loss\",factor=0.1,patience=8, mode=\"min\", min_lr=0.0001)\n\n]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:34:07.023886Z","iopub.execute_input":"2022-08-13T11:34:07.024177Z","iopub.status.idle":"2022-08-13T11:34:07.029164Z","shell.execute_reply.started":"2022-08-13T11:34:07.024148Z","shell.execute_reply":"2022-08-13T11:34:07.028064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Epochs=60\n\ncnn_model.fit_generator(augmented_training_data, \n                        epochs=Epochs,\n                        steps_per_epoch=29400 // Batch_size, #'29400' is 70% of total training data\n                        validation_data=validation_data,\n                        validation_steps=12600 // Batch_size, #'12600' is 30% of total training data\n                        callbacks=callbacks\n               )\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T11:34:20.301753Z","iopub.execute_input":"2022-08-13T11:34:20.302753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Prediction and Submission**","metadata":{}},{"cell_type":"code","source":"prediction = cnn_model.predict(test_x)\nprediction = prediction.tolist()\nlabels=[]\n\n#Generating CSV file for submission\n\nfor i in range(28000):\n    max_value = np.max(prediction[i])\n    labels.append(prediction[i].index(max_value))\n    \nsubmission = pd.DataFrame({\"ImageId\": list(range(1,len(test_x)+1)),\n                         \"Label\": labels})\n\nsubmission.to_csv(\"predictions_9.csv\", index=False)\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-03-04T11:47:40.499923Z","iopub.execute_input":"2022-03-04T11:47:40.500445Z","iopub.status.idle":"2022-03-04T11:47:42.319643Z","shell.execute_reply.started":"2022-03-04T11:47:40.500408Z","shell.execute_reply":"2022-03-04T11:47:42.318917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}