{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T17:20:05.966499Z","iopub.execute_input":"2022-08-13T17:20:05.967039Z","iopub.status.idle":"2022-08-13T17:20:05.977474Z","shell.execute_reply.started":"2022-08-13T17:20:05.966942Z","shell.execute_reply":"2022-08-13T17:20:05.976432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sns\n\n\nprint('Tensorflow Version {}'.format(tf.__version__))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:05.983550Z","iopub.execute_input":"2022-08-13T17:20:05.984097Z","iopub.status.idle":"2022-08-13T17:20:13.276842Z","shell.execute_reply.started":"2022-08-13T17:20:05.984022Z","shell.execute_reply":"2022-08-13T17:20:13.275637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ntrain_dataset = pd.read_csv('/kaggle/input/digit-recognizer/train.csv')\ntest_dataset = pd.read_csv('/kaggle/input/digit-recognizer/test.csv')\n\nprint('Total size of the train dataset is {}'.format(train_dataset.shape))\nprint('Total size of the test dataset is {}'.format(test_dataset.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:13.279018Z","iopub.execute_input":"2022-08-13T17:20:13.279331Z","iopub.status.idle":"2022-08-13T17:20:18.982337Z","shell.execute_reply.started":"2022-08-13T17:20:13.279301Z","shell.execute_reply":"2022-08-13T17:20:18.981273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_dataset['label']\n\nX_train = train_dataset.drop(['label'], axis=1)\n\n\nprint('Size of X_train {}'.format(X_train.shape))\nprint('Size of y_train {}'.format(y_train.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:18.984022Z","iopub.execute_input":"2022-08-13T17:20:18.984429Z","iopub.status.idle":"2022-08-13T17:20:19.119548Z","shell.execute_reply.started":"2022-08-13T17:20:18.984395Z","shell.execute_reply":"2022-08-13T17:20:19.118715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.isna().any().describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.121202Z","iopub.execute_input":"2022-08-13T17:20:19.121764Z","iopub.status.idle":"2022-08-13T17:20:19.162477Z","shell.execute_reply.started":"2022-08-13T17:20:19.121722Z","shell.execute_reply":"2022-08-13T17:20:19.161353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.isna().any().describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.166872Z","iopub.execute_input":"2022-08-13T17:20:19.167221Z","iopub.status.idle":"2022-08-13T17:20:19.183993Z","shell.execute_reply.started":"2022-08-13T17:20:19.167186Z","shell.execute_reply":"2022-08-13T17:20:19.182980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalization of the data\nX_train = X_train/255.0\ntest_dataset = test_dataset/255.0","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.186168Z","iopub.execute_input":"2022-08-13T17:20:19.186679Z","iopub.status.idle":"2022-08-13T17:20:19.300909Z","shell.execute_reply.started":"2022-08-13T17:20:19.186630Z","shell.execute_reply":"2022-08-13T17:20:19.299920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reshaping of the dataset\nX_train = X_train.values.reshape(-1, 28, 28, 1)\ntest_dataset = test_dataset.values.reshape(-1,28,28,1)\n\nprint('Shape of the train dataset is {}'.format(X_train.shape))\nprint('Shape of the test dataset is {}'.format(test_dataset.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.301906Z","iopub.execute_input":"2022-08-13T17:20:19.302188Z","iopub.status.idle":"2022-08-13T17:20:19.309447Z","shell.execute_reply.started":"2022-08-13T17:20:19.302161Z","shell.execute_reply":"2022-08-13T17:20:19.308147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\ny_train = to_categorical(y_train, num_classes=10)\n\nprint('Shape of y_train after one hot encoding {}'.format(y_train.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.311095Z","iopub.execute_input":"2022-08-13T17:20:19.311791Z","iopub.status.idle":"2022-08-13T17:20:19.326349Z","shell.execute_reply.started":"2022-08-13T17:20:19.311756Z","shell.execute_reply":"2022-08-13T17:20:19.325282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val, y_train, y_val = train_test_split(X_train, y_train, random_state=42, shuffle=True, test_size=0.2)\n\nprint('Size of train dataset is {}'.format(x_train.shape))\nprint('Size of the val dataset is {}'.format(x_val.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.327768Z","iopub.execute_input":"2022-08-13T17:20:19.328108Z","iopub.status.idle":"2022-08-13T17:20:19.738253Z","shell.execute_reply.started":"2022-08-13T17:20:19.328076Z","shell.execute_reply":"2022-08-13T17:20:19.737381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX = train_dataset.drop(['label'], axis=1, inplace=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.739798Z","iopub.execute_input":"2022-08-13T17:20:19.740116Z","iopub.status.idle":"2022-08-13T17:20:19.863118Z","shell.execute_reply.started":"2022-08-13T17:20:19.740085Z","shell.execute_reply":"2022-08-13T17:20:19.862014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nplt.style.use('ggplot')\nfor i in  range(20)  :\n    plt.subplot(4,5,i+1)\n    plt.imshow(X.values[ np.random.randint(1,X.shape[0])].reshape(28,28))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:19.864688Z","iopub.execute_input":"2022-08-13T17:20:19.865008Z","iopub.status.idle":"2022-08-13T17:20:21.639978Z","shell.execute_reply.started":"2022-08-13T17:20:19.864976Z","shell.execute_reply":"2022-08-13T17:20:21.638862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization of training set\nimport random\ni = random.randint(10,5000)\nplt.imshow(x_train[i][:,:,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:21.641443Z","iopub.execute_input":"2022-08-13T17:20:21.641890Z","iopub.status.idle":"2022-08-13T17:20:21.791210Z","shell.execute_reply.started":"2022-08-13T17:20:21.641846Z","shell.execute_reply":"2022-08-13T17:20:21.790015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting up model parameters\ninput_shape = (28,28,1)\nbatch_size = 64\nnum_classes = y_train.shape[1]\nepochs = 20","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:21.792715Z","iopub.execute_input":"2022-08-13T17:20:21.793133Z","iopub.status.idle":"2022-08-13T17:20:21.798635Z","shell.execute_reply.started":"2022-08-13T17:20:21.793092Z","shell.execute_reply":"2022-08-13T17:20:21.797389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CNN model\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.Conv2D(32, (5,5), padding ='same', activation='relu', input_shape=input_shape),\n    tf.keras.layers.Conv2D(32, (5,5), padding ='same', activation='relu'),\n    tf.keras.layers.MaxPool2D(),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Conv2D(64, (3,3), padding='same', activation='relu'),\n    tf.keras.layers.Conv2D(64, (3,3), padding='same', activation='relu'),\n    tf.keras.layers.MaxPool2D(),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Dense(num_classes, activation='softmax')   \n    \n])\n\nmodel.compile (optimizer=tf.keras.optimizers.RMSprop(epsilon=1e-08), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:21.800035Z","iopub.execute_input":"2022-08-13T17:20:21.800453Z","iopub.status.idle":"2022-08-13T17:20:21.974706Z","shell.execute_reply.started":"2022-08-13T17:20:21.800398Z","shell.execute_reply":"2022-08-13T17:20:21.973801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class myCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs={}):\n        if(logs.get('acc') is not None and logs.get('acc')> 0.995):\n            print(\"\\n Reached 99.5% of accuracy!!\")\n            self.model.stop_training=True\n            \ncallback= myCallback()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:21.976102Z","iopub.execute_input":"2022-08-13T17:20:21.976418Z","iopub.status.idle":"2022-08-13T17:20:21.982292Z","shell.execute_reply.started":"2022-08-13T17:20:21.976386Z","shell.execute_reply":"2022-08-13T17:20:21.981388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\ndatagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=False,  # apply ZCA whitening\n        rotation_range=10,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip=False,  # randomly flip images\n        vertical_flip=False)\n\ndatagen.fit(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:21.983731Z","iopub.execute_input":"2022-08-13T17:20:21.984060Z","iopub.status.idle":"2022-08-13T17:20:22.114424Z","shell.execute_reply.started":"2022-08-13T17:20:21.984020Z","shell.execute_reply":"2022-08-13T17:20:22.113533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(datagen.flow(x_train, y_train, batch_size = batch_size),\n                  epochs=epochs,\n                  validation_data= (x_val, y_val),\n                  callbacks = [callback])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:20:22.115594Z","iopub.execute_input":"2022-08-13T17:20:22.116139Z","iopub.status.idle":"2022-08-13T18:03:16.848806Z","shell.execute_reply.started":"2022-08-13T17:20:22.116089Z","shell.execute_reply":"2022-08-13T18:03:16.847842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:16.850099Z","iopub.execute_input":"2022-08-13T18:03:16.850538Z","iopub.status.idle":"2022-08-13T18:03:16.856652Z","shell.execute_reply.started":"2022-08-13T18:03:16.850508Z","shell.execute_reply":"2022-08-13T18:03:16.855689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(25,5))\n\n#Loss Curve\nax[0].plot(history.history['loss'], color='r', label='Training Loss')\nax[0].plot(history.history['val_loss'], color='b', label='Validation Loss')\nax[0].legend(loc='best')\nax[0].set_title('Loss curve for Training and validation DataSet', color='k', size=15)\n\n#Accuracy Curve\nax[1].plot(history.history['accuracy'], color='r', label='Training Accuracy')\nax[1].plot(history.history['val_accuracy'], color='b', label='Validation Accuracy')\nax[1].legend(loc='best')\nax[1].set_title('Accuracy curve for Training and Validation Dataset', size=15)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:16.857936Z","iopub.execute_input":"2022-08-13T18:03:16.858448Z","iopub.status.idle":"2022-08-13T18:03:17.284890Z","shell.execute_reply.started":"2022-08-13T18:03:16.858407Z","shell.execute_reply":"2022-08-13T18:03:17.283935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val_predict = model.predict(x_val)\n\ny_val_predict_class = np.argmax(y_val_predict, axis=1)\n\n\ny_val_true_class = np.argmax(y_val, axis=1)\n\n\nconfusion_matrix = tf.math.confusion_matrix(y_val_true_class, y_val_predict_class)\n\n\n# heatmap for confusion matrix\nplt.figure(figsize=(10,8))\nsns.heatmap(confusion_matrix, annot=True, fmt='g')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:17.286294Z","iopub.execute_input":"2022-08-13T18:03:17.286655Z","iopub.status.idle":"2022-08-13T18:03:25.377401Z","shell.execute_reply.started":"2022-08-13T18:03:17.286595Z","shell.execute_reply":"2022-08-13T18:03:25.376233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_dataset['label']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:25.378987Z","iopub.execute_input":"2022-08-13T18:03:25.379668Z","iopub.status.idle":"2022-08-13T18:03:25.385113Z","shell.execute_reply.started":"2022-08-13T18:03:25.379598Z","shell.execute_reply":"2022-08-13T18:03:25.383756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\nplt.pie(y.value_counts(),labels=list(y.value_counts().index),autopct ='%1.2f%%' ,\n        labeldistance = 1.1,explode = [0.05 for i in range(len(y.value_counts()))] )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:25.387697Z","iopub.execute_input":"2022-08-13T18:03:25.388629Z","iopub.status.idle":"2022-08-13T18:03:25.607845Z","shell.execute_reply.started":"2022-08-13T18:03:25.388558Z","shell.execute_reply":"2022-08-13T18:03:25.606490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict result of test dataset\n\npredict_test = model.predict(test_dataset)\nresults = np.argmax(predict_test, axis=1)\n\n\nprint(results)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:25.609374Z","iopub.execute_input":"2022-08-13T18:03:25.609866Z","iopub.status.idle":"2022-08-13T18:03:50.096641Z","shell.execute_reply.started":"2022-08-13T18:03:25.609827Z","shell.execute_reply":"2022-08-13T18:03:50.095303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting to csv for submission\nresults = pd.Series(results, name='Label')\n\nfinal_result = pd.concat([pd.Series(range(1, test_dataset.shape[0]+1), name='ImageId'), results], axis=1)\n\n\nfinal_result.to_csv('CNN_digit_classification_TensorFlow.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:03:50.100549Z","iopub.execute_input":"2022-08-13T18:03:50.101906Z","iopub.status.idle":"2022-08-13T18:03:50.148808Z","shell.execute_reply.started":"2022-08-13T18:03:50.101837Z","shell.execute_reply":"2022-08-13T18:03:50.147844Z"},"trusted":true},"execution_count":null,"outputs":[]}]}