{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"markdown","source":"**This is the first notebook which I created for this competition and was able to score 0.05128**\nSteps done:\n1. Trained a simple MNIST classifier \n2. Took the cleaned dataset from https://www.kaggle.com/remekkinas/ultramnistblack Thanks to [Remek Kinas](https://www.kaggle.com/remekkinas)\n3. Added contour detection algorithm(Image processing) to extract the numbers from the image.\n4. The detected contours were passed to the trained MNIST classifier to get the output and sum up the output\n5. Logged all the sum in the submission format.\n\nSpecial Thanks to the below person as the idea of cleaning the data was pretty awesome\n\n[LUKASZ BORECKI](https://www.kaggle.com/lukaszborecki) for this notebook:> https://www.kaggle.com/lukaszborecki/digit-cleaner-concept\n\nFor running end-to-end please refer to last section named **Whole inference pipeline end-to-end**\n\n**Notebook Time-line**\n\n**1st iteration:> 0.05128 score**\n\n**2nd iteration:> 0.37557 score**\n\n# Finally, Please Upvote my notebook if you found useful :)","metadata":{}},{"cell_type":"markdown","source":"# Train a simple MNIST model","metadata":{}},{"cell_type":"code","source":"# Training a MNIST model from TF-HUB\nimport tensorflow as tf\nimport tensorflow_datasets as tfds\n\n(ds_train, ds_test), ds_info = tfds.load(\n    'mnist',\n    split=['train', 'test'],\n    shuffle_files=True,\n    as_supervised=True,\n    with_info=True,\n)\n\ndef normalize_img(image, label):\n  \"\"\"Normalizes images: `uint8` -> `float32`.\"\"\"\n  return tf.cast(image, tf.float32) / 255., label\n\nds_train = ds_train.map(\n    normalize_img, num_parallel_calls=tf.data.AUTOTUNE)\nds_train = ds_train.cache()\nds_train = ds_train.shuffle(ds_info.splits['train'].num_examples)\nds_train = ds_train.batch(128)\nds_train = ds_train.prefetch(tf.data.AUTOTUNE)\n\nds_test = ds_test.map(\n    normalize_img, num_parallel_calls=tf.data.AUTOTUNE)\nds_test = ds_test.batch(128)\nds_test = ds_test.cache()\nds_test = ds_test.prefetch(tf.data.AUTOTUNE)\n\nmodel = tf.keras.models.Sequential([\n  tf.keras.layers.Flatten(input_shape=(28, 28)),\n  tf.keras.layers.Dense(128, activation='relu'),\n  tf.keras.layers.Dense(10)\n])\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(0.001),\n    loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n    metrics=[tf.keras.metrics.SparseCategoricalAccuracy()],\n)\n\nmodel.fit(\n    ds_train,\n    epochs=6,\n    validation_data=ds_test,\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:42:29.725619Z","iopub.execute_input":"2022-03-11T09:42:29.72592Z","iopub.status.idle":"2022-03-11T09:42:31.241978Z","shell.execute_reply.started":"2022-03-11T09:42:29.725866Z","shell.execute_reply":"2022-03-11T09:42:31.239052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretrained new Mnist model","metadata":{}},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.callbacks import ReduceLROnPlateau\nfrom sklearn.model_selection import train_test_split\nnum_classes = 10\ninput_shape = (28, 28, 1)\nmodel = Sequential()\nmodel.add(Conv2D(32, kernel_size=(3, 3),activation='relu',kernel_initializer='he_normal',input_shape=input_shape))\nmodel.add(Conv2D(32, kernel_size=(3, 3),activation='relu',kernel_initializer='he_normal'))\nmodel.add(MaxPool2D((2, 2)))\nmodel.add(Dropout(0.20))\nmodel.add(Conv2D(64, (3, 3), activation='relu',padding='same',kernel_initializer='he_normal'))\nmodel.add(Conv2D(64, (3, 3), activation='relu',padding='same',kernel_initializer='he_normal'))\nmodel.add(MaxPool2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(128, (3, 3), activation='relu',padding='same',kernel_initializer='he_normal'))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.25))\nmodel.add(Dense(num_classes, activation='softmax'))\n\nmodel.load_weights(\"../input/my-model-1h5/my_model_1.h5\")# loading pre-trained weights\n# model.summary()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:31.151673Z","iopub.execute_input":"2022-03-11T09:43:31.152078Z","iopub.status.idle":"2022-03-11T09:43:38.93858Z","shell.execute_reply.started":"2022-03-11T09:43:31.151949Z","shell.execute_reply":"2022-03-11T09:43:38.937908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Processing","metadata":{}},{"cell_type":"code","source":"!pip install imutils -q","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:38.940612Z","iopub.execute_input":"2022-03-11T09:43:38.941163Z","iopub.status.idle":"2022-03-11T09:43:52.245569Z","shell.execute_reply.started":"2022-03-11T09:43:38.941122Z","shell.execute_reply":"2022-03-11T09:43:52.244296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport imutils\nfrom imutils import contours","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:52.247418Z","iopub.execute_input":"2022-03-11T09:43:52.247734Z","iopub.status.idle":"2022-03-11T09:43:52.596367Z","shell.execute_reply.started":"2022-03-11T09:43:52.247698Z","shell.execute_reply":"2022-03-11T09:43:52.595377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = cv2.imread(\"../input/ultramnistblack/train/aejalacjek.jpeg\", 0)\nimage_oryg = cv2.imread(\"../input/ultra-mnist/train/aejalacjek.jpeg\", 0)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:52.600432Z","iopub.execute_input":"2022-03-11T09:43:52.601185Z","iopub.status.idle":"2022-03-11T09:43:52.734736Z","shell.execute_reply.started":"2022-03-11T09:43:52.60113Z","shell.execute_reply":"2022-03-11T09:43:52.733838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_list = []\n\ncnts = cv2.findContours(image.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) #image, mask\ncnts = imutils.grab_contours(cnts)\ncnts = contours.sort_contours(cnts)[0]\nprint(f'Found {len(cnts)} contours / numbers')\nbacktorgb = cv2.cvtColor(image.astype('float32'), cv2.COLOR_GRAY2RGB)\n\nfor (i, c) in enumerate(cnts):\n    (x, y, w, h) = cv2.boundingRect(c)\n    bbox_list.append([x, y, w, h])\n    cv2.rectangle(backtorgb, (x,y), (x+w, y+h), (255,0,0), 5)\n\nprint(f'BBoxes coordinates: {bbox_list}')","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:52.736008Z","iopub.execute_input":"2022-03-11T09:43:52.736927Z","iopub.status.idle":"2022-03-11T09:43:52.849312Z","shell.execute_reply.started":"2022-03-11T09:43:52.736877Z","shell.execute_reply":"2022-03-11T09:43:52.84868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count=0\nsum_val=0\nfor bbox in bbox_list:\n    bbox_img = image[bbox[1]-15:bbox[1]+bbox[3]+15, bbox[0]-15:bbox[0]+bbox[2]+15]\n    \n    if bbox_img.shape[0] > 100:\n        count+= 1\n        bbox_resized = cv2.resize(bbox_img, (28, 28), interpolation = cv2.INTER_AREA)\n        img2 = np.expand_dims(bbox_resized, axis=-1)\n        img2 = np.expand_dims(img2, axis=0)\n        \n        pred= model.predict(img2)\n        \n        print(\"pred:>\",np.argmax(pred))\n        sum_val+= int(np.argmax(pred))\n        \n        if sum_val > 27:\n            sum_val=27\n        plt.imshow(bbox_resized,cmap='gray')\n        plt.show()\n#         display(Img.fromarray((bbox_resized).astype(np.uint8)))\n#         break\nprint(\"final count:> \",count,sum_val)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:52.850563Z","iopub.execute_input":"2022-03-11T09:43:52.850998Z","iopub.status.idle":"2022-03-11T09:43:54.393682Z","shell.execute_reply.started":"2022-03-11T09:43:52.850966Z","shell.execute_reply":"2022-03-11T09:43:54.392833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# writing to CSV\ncsv_path= r'../input/ultra-mnist/sample_submission.csv'\ncsv= pd.read_csv(csv_path)\ncsv.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:42:31.254973Z","iopub.status.idle":"2022-03-11T09:42:31.25605Z","shell.execute_reply.started":"2022-03-11T09:42:31.255777Z","shell.execute_reply":"2022-03-11T09:42:31.255803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv.at[0,'digit_sum'] = 222\ncsv.at[0,'id']= 'ash'","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:42:31.257182Z","iopub.status.idle":"2022-03-11T09:42:31.258024Z","shell.execute_reply.started":"2022-03-11T09:42:31.257746Z","shell.execute_reply":"2022-03-11T09:42:31.257772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:42:31.259348Z","iopub.status.idle":"2022-03-11T09:42:31.259755Z","shell.execute_reply.started":"2022-03-11T09:42:31.259533Z","shell.execute_reply":"2022-03-11T09:42:31.259557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Whole inference pipeline end-to-end","metadata":{}},{"cell_type":"code","source":"from imutils import paths\ntest_data= r'../input/ultramnistblack/test'\n# writing to the csv for submission\ncsv_path= r'../input/ultra-mnist/sample_submission.csv'\ndf= pd.read_csv(csv_path)\n\nfor idx,img_path in enumerate(paths.list_files(test_data,'.jpeg')):\n    #read the image\n    image= cv2.imread(img_path,0)\n#     plt.imshow(image, cmap='gray')\n#     plt.show()\n    #find contours from the image------------------------------------------->\n    bbox_list = []\n    cnts = cv2.findContours(image.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) #image, mask\n    cnts = imutils.grab_contours(cnts)\n    cnts = contours.sort_contours(cnts)[0]\n#     print(f'Found {len(cnts)} contours / numbers')\n    \n    for (i, c) in enumerate(cnts):\n        (x, y, w, h) = cv2.boundingRect(c)\n        bbox_list.append([x, y, w, h])\n        \n    #filter the boxes------------------------------------------------------>\n    count=0\n    sum_val=0\n    for bbox in bbox_list:\n        bbox_img = image[bbox[1]-15:bbox[1]+bbox[3]+15, bbox[0]-15:bbox[0]+bbox[2]+15]\n    \n        if bbox_img.shape[0] > 100:\n            count+= 1\n            bbox_resized = cv2.resize(bbox_img, (28, 28), interpolation = cv2.INTER_AREA)\n            img2 = np.expand_dims(bbox_resized, axis=-1)\n            img2 = np.expand_dims(img2, axis=0)\n\n            pred= model.predict(img2)\n\n#             print(\"pred:>\",np.argmax(pred))\n            sum_val+= int(np.argmax(pred))\n\n            if sum_val > 27:\n                sum_val=27\n#             plt.imshow(bbox_resized,cmap='gray')\n#             plt.show()\n            \n    df.at[idx,'digit_sum']= sum_val\n    df.at[idx,'id']= str(img_path.split('/')[-1].split('.')[0])\n    print(str(idx)+' done !!')\n#     print(\"count:> \",count,img_path.split('/')[-1])\n#     print(\"sum:> \",sum_val)\n#     break\ndf.to_csv('submission.csv', index=False)\ndf.head()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-11T09:43:56.585753Z","iopub.execute_input":"2022-03-11T09:43:56.586104Z","iopub.status.idle":"2022-03-11T12:03:41.200933Z","shell.execute_reply.started":"2022-03-11T09:43:56.586062Z","shell.execute_reply":"2022-03-11T12:03:41.199834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}