{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading data and libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2 as cv\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras.layers import *\nfrom keras.models import Model\n\nfrom matplotlib import pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-02T22:32:52.754034Z","iopub.execute_input":"2022-04-02T22:32:52.754630Z","iopub.status.idle":"2022-04-02T22:32:59.483421Z","shell.execute_reply.started":"2022-04-02T22:32:52.754530Z","shell.execute_reply":"2022-04-02T22:32:59.482430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/ultra-mnist/train.csv')\ntest_df = pd.read_csv('../input/ultra-mnist/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:32:59.485109Z","iopub.execute_input":"2022-04-02T22:32:59.485358Z","iopub.status.idle":"2022-04-02T22:32:59.549950Z","shell.execute_reply.started":"2022-04-02T22:32:59.485327Z","shell.execute_reply":"2022-04-02T22:32:59.548993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining generators and auxiliary functions","metadata":{}},{"cell_type":"code","source":"def conv2D(var, kernel, stride):\n\n    ny, nx = var.shape\n    ky, kx = kernel.shape\n    result = 0\n    \n    for ii in range(ky * kx):\n        yi, xi = divmod(ii, kx)\n        result += var[yi:ny - ky + yi + 1, xi:nx - kx + xi + 1] * kernel[yi, xi]\n        \n    if stride > 1:\n        result = result[::stride, ::stride]\n        \n    return result","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:32:59.551714Z","iopub.execute_input":"2022-04-02T22:32:59.552315Z","iopub.status.idle":"2022-04-02T22:32:59.559314Z","shell.execute_reply.started":"2022-04-02T22:32:59.552273Z","shell.execute_reply":"2022-04-02T22:32:59.558496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def my_conv(var, kernel_size=3, stride=2, threshold=2):\n    \n    h = np.zeros((kernel_size, kernel_size))\n    h[kernel_size // 2 - 1, :] = -1\n    h[kernel_size // 2 + 1, :] = 1\n    v = np.zeros((kernel_size, kernel_size))\n    v[:, kernel_size // 2 - 1] = -1\n    v[:, kernel_size // 2 + 1] = 1\n    \n    result = (np.abs(conv2D(var, h, stride)) > threshold) | (np.abs(conv2D(var, v, stride)) > threshold)\n    \n    return result","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:32:59.561072Z","iopub.execute_input":"2022-04-02T22:32:59.561321Z","iopub.status.idle":"2022-04-02T22:32:59.574711Z","shell.execute_reply.started":"2022-04-02T22:32:59.561294Z","shell.execute_reply":"2022-04-02T22:32:59.573828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padding(img, d):\n    \n    h, w = img.shape\n    new_img = np.vstack([np.zeros((d, w), 'uint8'), img, np.zeros((d, w), 'uint8')])\n    left, right = (28 - w) // 2, 28 - w - (28 - w) // 2\n    result = np.hstack([np.zeros((h + 2 * d, d), 'uint8'), new_img, np.zeros((h + 2 * d, d), 'uint8')])\n    \n    return result","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:32:59.576954Z","iopub.execute_input":"2022-04-02T22:32:59.577785Z","iopub.status.idle":"2022-04-02T22:32:59.588591Z","shell.execute_reply.started":"2022-04-02T22:32:59.577716Z","shell.execute_reply":"2022-04-02T22:32:59.587764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UltraMnistDataGenTrain(tf.keras.utils.Sequence):\n    \n    def __init__(self, steps, img_size):\n        self.steps = steps\n        self.img_size = img_size\n        \n    def __getitem__(self, idx):\n\n        batch_img = np.zeros((9, self.img_size, self.img_size))\n\n        sample = train_df.sample()\n        name_img = sample['id'].to_numpy()[0]\n        img = (cv.imread(f'../input/ultra-mnist/train/{name_img}.jpeg', 0) > 100)\n        label_img = sample['digit_sum'].to_numpy()[0]\n        img = np.vstack([np.hstack([padding(my_conv(img), 24), np.zeros((2047, 1))]), np.zeros((1, 2048))])\n        \n        for i in range(3):\n            for j in range(3):\n                    crop_img = img[i * self.img_size // 2:(i + 2) * self.img_size // 2, \n                                   j * self.img_size // 2:(j + 2) * self.img_size // 2]\n                    batch_img[i * 3 + j] = crop_img\n\n        return batch_img, (img, label_img)\n    \n    def __len__(self):\n        return self.steps\n    \n    def on_epoch_end(self):\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:32:59.590095Z","iopub.execute_input":"2022-04-02T22:32:59.590488Z","iopub.status.idle":"2022-04-02T22:33:00.024057Z","shell.execute_reply.started":"2022-04-02T22:32:59.590443Z","shell.execute_reply":"2022-04-02T22:33:00.023098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_img = test_df.iloc[0, 0]\ncv.imread(f'../input/ultra-mnist/test/{name_img}.jpeg', 0) > 0","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.025213Z","iopub.execute_input":"2022-04-02T22:33:00.025431Z","iopub.status.idle":"2022-04-02T22:33:00.130821Z","shell.execute_reply.started":"2022-04-02T22:33:00.025406Z","shell.execute_reply":"2022-04-02T22:33:00.129741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UltraMnistDataGenTest(tf.keras.utils.Sequence):\n    \n    def __init__(self, steps, img_size):\n        self.steps = steps\n        self.img_size = img_size\n        \n    def __getitem__(self, idx):\n\n        batch_img = np.zeros((9, self.img_size, self.img_size))\n\n        name_img = test_df.iloc[idx, 0]\n        img = (cv.imread(f'../input/ultra-mnist/test/{name_img}.jpeg', 0) > 100)\n        img = np.vstack([np.hstack([padding(my_conv(img), 24), np.zeros((2047, 1))]), np.zeros((1, 2048))])\n        \n        for i in range(3):\n            for j in range(3):\n                    crop_img = img[i * self.img_size // 2:(i + 2) * self.img_size // 2, \n                                   j * self.img_size // 2:(j + 2) * self.img_size // 2]\n                    batch_img[i * 3 + j] = crop_img\n\n        return batch_img\n    \n    def __len__(self):\n        return self.steps\n    \n    def on_epoch_end(self):\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.132043Z","iopub.execute_input":"2022-04-02T22:33:00.132265Z","iopub.status.idle":"2022-04-02T22:33:00.142687Z","shell.execute_reply.started":"2022-04-02T22:33:00.132239Z","shell.execute_reply":"2022-04-02T22:33:00.142026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building models","metadata":{}},{"cell_type":"markdown","source":"I decided to use two different YOLO models with different grid sizes (because big grid is good for big digits, and small one is good for small digits).","metadata":{}},{"cell_type":"code","source":"# general parameters of YOLO\nIMAGE_SIZE       = 1024\nGRID_SIZE_1      = 32\nGRID_SIZE_2      = 8\n\n# parameters that determine non-max supression\nCONF_THRESHOLD   = 0.8\nIOU_THRESHOLD    = 0.1\n\n# parameters that determine loss function\nCLASS_SCALE      = 2.\nLOC_SCALE        = 3.\nOBJ_SCALE        = 1.\nNOBJ_SCALE       = 1.\n\n# training parameters\nBATCH_SIZE       = 16\nSTEPS            = 100","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.143985Z","iopub.execute_input":"2022-04-02T22:33:00.144417Z","iopub.status.idle":"2022-04-02T22:33:00.159415Z","shell.execute_reply.started":"2022-04-02T22:33:00.144387Z","shell.execute_reply":"2022-04-02T22:33:00.158711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Source of architecture of this model: https://github.com/experiencor/keras-yolo2/blob/master/Yolo%20Step-by-Step.ipynb\n\nBut I make some changes. And the main change is output layer (because we don't need anchor boxes). ","metadata":{}},{"cell_type":"code","source":"def yolo_model(img_shape, grid_size):\n    \n    input_image = Input(shape=img_shape)\n\n    # Layer 1\n    x = Conv2D(16, (3,3), strides=(1,1), padding='same', name='conv_1', use_bias=False)(input_image)\n    x = BatchNormalization(name='norm_1')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 2\n    x = Conv2D(32, (3,3), strides=(1,1), padding='same', name='conv_2', use_bias=False)(x)\n    x = BatchNormalization(name='norm_2')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 3\n    x = Conv2D(64, (3,3), strides=(1,1), padding='same', name='conv_3', use_bias=False)(x)\n    x = BatchNormalization(name='norm_3')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    if grid_size < 32:\n        x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    # Layer 4\n    x = Conv2D(32, (1,1), strides=(1,1), padding='same', name='conv_4', use_bias=False)(x)\n    x = BatchNormalization(name='norm_4')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    if grid_size < 16:\n        x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 5\n    x = Conv2D(64, (3,3), strides=(1,1), padding='same', name='conv_5', use_bias=False)(x)\n    x = BatchNormalization(name='norm_5')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 6\n    x = Conv2D(128, (3,3), strides=(1,1), padding='same', name='conv_6', use_bias=False)(x)\n    x = BatchNormalization(name='norm_6')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 7\n    x = Conv2D(64, (1,1), strides=(1,1), padding='same', name='conv_7', use_bias=False)(x)\n    x = BatchNormalization(name='norm_7')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 8\n    x = Conv2D(128, (3,3), strides=(1,1), padding='same', name='conv_8', use_bias=False)(x)\n    x = BatchNormalization(name='norm_8')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 9\n    x = Conv2D(256, (3,3), strides=(1,1), padding='same', name='conv_9', use_bias=False)(x)\n    x = BatchNormalization(name='norm_9')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 10\n    x = Conv2D(128, (1,1), strides=(1,1), padding='same', name='conv_10', use_bias=False)(x)\n    x = BatchNormalization(name='norm_10')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 11\n    x = Conv2D(256, (3,3), strides=(1,1), padding='same', name='conv_11', use_bias=False)(x)\n    x = BatchNormalization(name='norm_11')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 12\n    x = Conv2D(128, (1,1), strides=(1,1), padding='same', name='conv_12', use_bias=False)(x)\n    x = BatchNormalization(name='norm_12')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 13\n    x = Conv2D(256, (3,3), strides=(1,1), padding='same', name='conv_13', use_bias=False)(x)\n    x = BatchNormalization(name='norm_13')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    skip_connection = x\n\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n\n    # Layer 14\n    x = Conv2D(512, (3,3), strides=(1,1), padding='same', name='conv_14', use_bias=False)(x)\n    x = BatchNormalization(name='norm_14')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 15\n    x = Conv2D(256, (1,1), strides=(1,1), padding='same', name='conv_15', use_bias=False)(x)\n    x = BatchNormalization(name='norm_15')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 16\n    x = Conv2D(512, (3,3), strides=(1,1), padding='same', name='conv_16', use_bias=False)(x)\n    x = BatchNormalization(name='norm_16')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 17\n    x = Conv2D(256, (1,1), strides=(1,1), padding='same', name='conv_17', use_bias=False)(x)\n    x = BatchNormalization(name='norm_17')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 18\n    x = Conv2D(512, (3,3), strides=(1,1), padding='same', name='conv_18', use_bias=False)(x)\n    x = BatchNormalization(name='norm_18')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 19\n    x = Conv2D(512, (3,3), strides=(1,1), padding='same', name='conv_19', use_bias=False)(x)\n    x = BatchNormalization(name='norm_19')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 20\n    x = Conv2D(512, (3,3), strides=(1,1), padding='same', name='conv_20', use_bias=False)(x)\n    x = BatchNormalization(name='norm_20')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 21\n    skip_connection = Conv2D(32, (1,1), strides=(1,1), padding='same', name='conv_21', use_bias=False)(skip_connection)\n    skip_connection = BatchNormalization(name='norm_21')(skip_connection)\n    skip_connection = LeakyReLU(alpha=0.1)(skip_connection)\n    skip_connection = Lambda(lambda x: tf.nn.space_to_depth(x, block_size=2))(skip_connection)\n\n    x = concatenate([skip_connection, x])\n\n    # Layer 22\n    x = Conv2D(32, (1,1), strides=(1,1), padding='same', name='conv_22', use_bias=False)(x)\n    x = BatchNormalization(name='norm_22')(x)\n    x = LeakyReLU(alpha=0.1)(x)\n\n    # Layer 23\n    output = Conv2D(15, (1,1), strides=(1,1), padding='same', name='conv_23')(x)\n    output = Lambda(lambda x: tf.concat([tf.nn.sigmoid(x[..., :1]), \n                                         tf.nn.leaky_relu(x[..., 1:5]), \n                                         tf.nn.softmax(x[..., 5:])], \n                                         axis=-1))(output)\n\n    model = Model(input_image, output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.162412Z","iopub.execute_input":"2022-04-02T22:33:00.162790Z","iopub.status.idle":"2022-04-02T22:33:00.206731Z","shell.execute_reply.started":"2022-04-02T22:33:00.162728Z","shell.execute_reply":"2022-04-02T22:33:00.206053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Source of loss function: https://arxiv.org/pdf/1506.02640.pdf","metadata":{}},{"cell_type":"code","source":"def yolo_loss(y_true, y_pred):\n    \n    true_obj = y_true[..., :1]\n    pred_obj = y_pred[..., :1]\n    \n    true_xy = y_true[..., 1:3]\n    pred_xy = y_pred[..., 1:3]\n    \n    true_wh = y_true[..., 3:5]\n    pred_wh = y_pred[..., 3:5]\n    \n    true_class = y_true[..., 5:]\n    pred_class = y_pred[..., 5:]\n    \n    # classification loss\n\n    class_loss = CLASS_SCALE * tf.reduce_sum(true_obj * tf.square(true_class - pred_class))\n    \n    # localization loss\n    \n    xy_loss = tf.reduce_sum(true_obj * tf.square(true_xy - pred_xy))\n    wh_loss = tf.reduce_sum(true_obj * tf.square(true_wh - pred_wh))\n    #tf.print(xy_loss, wh_loss)\n    loc_loss = LOC_SCALE * (xy_loss + wh_loss)\n    \n    # confidence loss\n    \n    obj_loss = tf.reduce_sum(true_obj * tf.square(true_obj - pred_obj))\n    nobj_loss = NOBJ_SCALE * tf.reduce_sum((1. - true_obj) * tf.square(true_obj - pred_obj))\n    #tf.print(obj_loss, nobj_loss)\n    conf_loss = obj_loss + nobj_loss\n    \n    # summary loss\n    \n    #tf.print('class:', class_loss, 'loc:', loc_loss, 'conf:', conf_loss)\n    loss = class_loss + loc_loss + conf_loss\n    \n    return loss","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.208122Z","iopub.execute_input":"2022-04-02T22:33:00.208568Z","iopub.status.idle":"2022-04-02T22:33:00.223187Z","shell.execute_reply.started":"2022-04-02T22:33:00.208527Z","shell.execute_reply":"2022-04-02T22:33:00.222314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this notebook I used already pre-trained model on generated dataset (for generating I used original MNIST). \n\nYou also can use this: https://www.kaggle.com/datasets/dmitrylessy/ultramnist-models\n\nGenerated dataset: https://www.kaggle.com/datasets/dmitrylessy/generated-ultra-mnist","metadata":{}},{"cell_type":"code","source":"model1 = yolo_model((IMAGE_SIZE, IMAGE_SIZE, 1), GRID_SIZE_1)\nmodel1.compile('adam', yolo_loss)\n\n#train_gen_1 = FlowFromDirectoryDataGen(STEPS, BATCH_SIZE, IMAGE_SIZE, GRID_SIZE_1)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.224914Z","iopub.execute_input":"2022-04-02T22:33:00.226925Z","iopub.status.idle":"2022-04-02T22:33:00.838523Z","shell.execute_reply.started":"2022-04-02T22:33:00.226881Z","shell.execute_reply":"2022-04-02T22:33:00.837770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.load_weights(f'../input/ultramnist-models/yolo_weights_grid_{GRID_SIZE_1}.h5')","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:00.839862Z","iopub.execute_input":"2022-04-02T22:33:00.840169Z","iopub.status.idle":"2022-04-02T22:33:01.735688Z","shell.execute_reply.started":"2022-04-02T22:33:00.840129Z","shell.execute_reply":"2022-04-02T22:33:01.734698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model1.fit(train_gen_1, epochs=50, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:01.737157Z","iopub.execute_input":"2022-04-02T22:33:01.737401Z","iopub.status.idle":"2022-04-02T22:33:01.741451Z","shell.execute_reply.started":"2022-04-02T22:33:01.737374Z","shell.execute_reply":"2022-04-02T22:33:01.740521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2 = yolo_model((IMAGE_SIZE, IMAGE_SIZE, 1), GRID_SIZE_2)\nmodel2.compile('adam', yolo_loss)\n\n#train_gen_2 = FlowFromDirectoryDataGen(STEPS, BATCH_SIZE, IMAGE_SIZE, GRID_SIZE_2)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:01.742659Z","iopub.execute_input":"2022-04-02T22:33:01.742916Z","iopub.status.idle":"2022-04-02T22:33:02.255647Z","shell.execute_reply.started":"2022-04-02T22:33:01.742885Z","shell.execute_reply":"2022-04-02T22:33:02.254553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2.load_weights(f'../input/ultramnist-models/yolo_weights_grid_{GRID_SIZE_2}.h5')","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:02.257285Z","iopub.execute_input":"2022-04-02T22:33:02.257559Z","iopub.status.idle":"2022-04-02T22:33:03.036048Z","shell.execute_reply.started":"2022-04-02T22:33:02.257528Z","shell.execute_reply":"2022-04-02T22:33:03.035136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model2.fit(train_gen_2, epochs=100, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.037347Z","iopub.execute_input":"2022-04-02T22:33:03.038965Z","iopub.status.idle":"2022-04-02T22:33:03.043945Z","shell.execute_reply.started":"2022-04-02T22:33:03.038926Z","shell.execute_reply":"2022-04-02T22:33:03.043018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Non-max supression and evaluation our models","metadata":{}},{"cell_type":"markdown","source":"I made some changes in standard Intersection Over Union function, because we know that digits cannot intesect.","metadata":{}},{"cell_type":"code","source":"def IoU(box1, box2):\n    \n    box1_x1, box1_y1, box1_x2, box1_y2 = box1\n    box2_x1, box2_y1, box2_x2, box2_y2 = box2\n\n    inter_width = min(box1_x2, box2_x2)  -  max(box1_x1, box2_x1) \n    inter_height =  min(box1_y2, box2_y2)  - max(box1_y1, box2_y1) \n    inter_area = max(inter_height, 0) * max(inter_width, 0)\n    \n    box1_area = (box1_x1 - box1_x2) * (box1_y1 - box1_y2)\n    box2_area = (box2_x1 - box2_x2) * (box2_y1 - box2_y2)\n    union_area = box1_area + box2_area - inter_area\n    \n    iou = inter_area / min(box1_area, box2_area)\n    \n    return iou","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.045241Z","iopub.execute_input":"2022-04-02T22:33:03.045790Z","iopub.status.idle":"2022-04-02T22:33:03.054628Z","shell.execute_reply.started":"2022-04-02T22:33:03.045728Z","shell.execute_reply":"2022-04-02T22:33:03.053871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This function convert prediction from YOLO model to more appropriate shape, and then filter them by confidence and intersections.","metadata":{}},{"cell_type":"code","source":"def non_max_supression(data, grid_size, batch_size=BATCH_SIZE):\n    \n    filtered_data = data.copy()\n    \n    cell_x = tf.reshape(tf.tile(tf.range(grid_size, dtype='float32'), [grid_size]),[1, grid_size, grid_size, 1])\n    cell_y = tf.transpose(cell_x, (0,2,1,3))\n    cell_grid = tf.tile(tf.concat([cell_x,cell_y], -1), [batch_size, 1, 1, 1])\n    \n    xy = (data[..., 1:3] + cell_grid) * (IMAGE_SIZE // grid_size)\n    wh = tf.square(data[..., 3:5]) * (IMAGE_SIZE // grid_size)\n    \n    top = xy - wh / 2\n    bottom = xy + wh / 2\n    \n    filtered_data[..., 1:3] = top\n    filtered_data[..., 3:5] = bottom\n    \n    filtered_data = list(filtered_data)\n    \n    for i in range(len(filtered_data)):\n        filtered_data[i] = filtered_data[i][filtered_data[i][..., 0] > CONF_THRESHOLD]\n        filtered_data[i] = filtered_data[i][filtered_data[i][..., 0].argsort()]\n        \n        j = 0\n        while j < len(filtered_data[i]):\n            box1 = filtered_data[i][j, 1:5]\n            for k in range(j + 1, len(filtered_data[i])):\n                box2 = filtered_data[i][k, 1:5]\n                iou = IoU(box1, box2)\n                if iou > IOU_THRESHOLD:\n                    filtered_data[i] = np.delete(filtered_data[i], j, axis=0)\n                    j -= 1\n                    break\n            j += 1\n\n    return filtered_data","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.055733Z","iopub.execute_input":"2022-04-02T22:33:03.056101Z","iopub.status.idle":"2022-04-02T22:33:03.070410Z","shell.execute_reply.started":"2022-04-02T22:33:03.056063Z","shell.execute_reply":"2022-04-02T22:33:03.069510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This function unites outputs from two models to one prediction.","metadata":{}},{"cell_type":"code","source":"def convert_coords(data, img_size=1024):\n    all_labels = data.copy()\n    all_boxes = []\n    \n    for labels in all_labels:\n        for i in range(3):\n            for j in range(3):\n                boxes = labels[i * 3 + j].copy()\n\n                boxes[..., 1] = boxes[..., 1] + j * img_size // 2\n                boxes[..., 3] = boxes[..., 3] + j * img_size // 2\n                boxes[..., 2] = boxes[..., 2] + i * img_size // 2\n                boxes[..., 4] = boxes[..., 4] + i * img_size // 2\n\n                all_boxes.extend(list(boxes))\n    \n    all_boxes = np.array(all_boxes)\n    all_boxes = all_boxes.reshape(-1, 15)\n    all_boxes = all_boxes[all_boxes[..., 0].argsort()]\n    j = 0\n    while j < len(all_boxes):\n        box1 = all_boxes[j]\n        k = j + 1\n        while k < len(all_boxes):\n            box2 = all_boxes[k]\n            iou = IoU(box1[1:5], box2[1:5])\n            if iou > IOU_THRESHOLD:\n                s1 = (box1[4] - box1[2]) * (box1[3] - box1[1])\n                s2 = (box2[4] - box2[2]) * (box2[3] - box2[1])\n                if s1 <= s2 or box2[0] - box1[0] > 0.1:\n                    all_boxes = np.delete(all_boxes, j, axis=0)\n                    j -= 1\n                    break\n                else:\n                    all_boxes = np.delete(all_boxes, k, axis=0)\n                    k -= 1\n            k += 1\n        j += 1\n        \n    all_boxes = all_boxes[-5:]\n    \n    while sum(get_labels(all_boxes)) > 27:\n        if all_boxes[0][0] < 0.8:\n            all_boxes = all_boxes[1:]\n        else:\n            probs = all_boxes[:, 5:].max(axis=-1)\n            i = probs.argmin()\n            j = all_boxes[i, 5:].argmax(axis=-1)\n            all_boxes[i, 5 + j] = 0\n            all_boxes[i, 0] = 0\n            all_boxes = all_boxes[all_boxes[..., 0].argsort()]\n    \n    return all_boxes","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.071676Z","iopub.execute_input":"2022-04-02T22:33:03.072046Z","iopub.status.idle":"2022-04-02T22:33:03.092158Z","shell.execute_reply.started":"2022-04-02T22:33:03.072014Z","shell.execute_reply":"2022-04-02T22:33:03.091253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_corners(data):\n\n    arr = []\n    \n    for box in data:\n        arr.append(box[1:5].astype(int))\n    \n    return arr","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.093510Z","iopub.execute_input":"2022-04-02T22:33:03.094365Z","iopub.status.idle":"2022-04-02T22:33:03.106000Z","shell.execute_reply.started":"2022-04-02T22:33:03.094328Z","shell.execute_reply":"2022-04-02T22:33:03.105398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_labels(data):\n    \n    arr = []\n    \n    for box in data:\n        arr.append(box[5:].argmax())\n        \n    return arr","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.107220Z","iopub.execute_input":"2022-04-02T22:33:03.107432Z","iopub.status.idle":"2022-04-02T22:33:03.119840Z","shell.execute_reply.started":"2022-04-02T22:33:03.107406Z","shell.execute_reply":"2022-04-02T22:33:03.118847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_rects(img, coord):\n    \n    new_img = img.copy()\n    \n    for x1, y1, x2, y2 in coord:\n        cv.line(new_img, (x1, y1), (x1, y2), 1, 3)\n        cv.line(new_img, (x1, y1), (x2, y1), 1, 3)\n        cv.line(new_img, (x2, y2), (x1, y2), 1, 3)\n        cv.line(new_img, (x2, y2), (x2, y1), 1, 3)\n        \n    return new_img","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.120974Z","iopub.execute_input":"2022-04-02T22:33:03.121361Z","iopub.status.idle":"2022-04-02T22:33:03.139053Z","shell.execute_reply.started":"2022-04-02T22:33:03.121327Z","shell.execute_reply":"2022-04-02T22:33:03.138005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Example of performing out model on a single image:","metadata":{}},{"cell_type":"code","source":"ultra_gen = UltraMnistDataGenTrain(1, IMAGE_SIZE)\nx, y = ultra_gen[0]\npred1 = model1.predict(x)\npred1 = non_max_supression(pred1, GRID_SIZE_1, batch_size=9)\npred2 = model2.predict(x)\npred2 = non_max_supression(pred2, GRID_SIZE_2, batch_size=9)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:03.140232Z","iopub.execute_input":"2022-04-02T22:33:03.140836Z","iopub.status.idle":"2022-04-02T22:33:09.831064Z","shell.execute_reply.started":"2022-04-02T22:33:03.140803Z","shell.execute_reply":"2022-04-02T22:33:09.830256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(1, 1, figsize=(18, 18))\nimg = y[0]\ndata = convert_coords([pred1, pred2])\ncorners = get_corners(data)\npred_labels = get_labels(data)\nimg = draw_rects(img, corners)\nax.imshow(img)\nax.set_title('pred: ' + str(sum(pred_labels)) + ', real: ' + str(y[1]))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:09.832419Z","iopub.execute_input":"2022-04-02T22:33:09.832632Z","iopub.status.idle":"2022-04-02T22:33:10.966447Z","shell.execute_reply.started":"2022-04-02T22:33:09.832606Z","shell.execute_reply":"2022-04-02T22:33:10.965741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(3, 3, figsize=(18, 18))\nfor i in range(9):\n    img = x[i]\n    corners1 = get_corners(pred1[i])\n    pred_labels1 = get_labels(pred1[i])\n    corners2 = get_corners(pred2[i])\n    pred_labels2 = get_labels(pred2[i])\n    img = draw_rects(img, corners1 + corners2)\n    ax[i // 3][i % 3].imshow(img)\n    ax[i // 3][i % 3].set_title('pred1: ' + str(pred_labels1) + ' pred2: ' + str(pred_labels2))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:10.967581Z","iopub.execute_input":"2022-04-02T22:33:10.968169Z","iopub.status.idle":"2022-04-02T22:33:13.556418Z","shell.execute_reply.started":"2022-04-02T22:33:10.968136Z","shell.execute_reply":"2022-04-02T22:33:13.555799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can estimate accuracy on train set:","metadata":{}},{"cell_type":"code","source":"t = 0\nultra_gen = UltraMnistDataGenTrain(1, IMAGE_SIZE)\nfor i in range(10):\n    x, y = ultra_gen[i]\n    pred1 = model1.predict(x)\n    pred1 = non_max_supression(pred1, GRID_SIZE_1, batch_size=9)\n    pred2 = model2.predict(x)\n    pred2 = non_max_supression(pred2, GRID_SIZE_2, batch_size=9)\n    data = convert_coords([pred1, pred2])\n    pred_labels = get_labels(data)\n    if sum(pred_labels) == y[1]:\n        t += 1\nprint(f'Train accuracy: {t / 100}%')","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:33:13.557617Z","iopub.execute_input":"2022-04-02T22:33:13.558058Z","iopub.status.idle":"2022-04-02T22:34:04.390434Z","shell.execute_reply.started":"2022-04-02T22:33:13.558026Z","shell.execute_reply":"2022-04-02T22:34:04.389487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test_gen = UltraMnistDataGenTest(STEPS, IMAGE_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:04:33.953161Z","iopub.execute_input":"2022-04-02T22:04:33.953441Z","iopub.status.idle":"2022-04-02T22:04:33.957472Z","shell.execute_reply.started":"2022-04-02T22:04:33.953408Z","shell.execute_reply":"2022-04-02T22:04:33.956599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(28000):\n    x = test_gen[i]\n    pred1 = model1.predict(x)\n    pred1 = non_max_supression(pred1, GRID_SIZE_1, batch_size=9)\n    pred2 = model2.predict(x)\n    pred2 = non_max_supression(pred2, GRID_SIZE_2, batch_size=9)\n    data = convert_coords([pred1, pred2])\n    pred_labels = get_labels(data)\n    pred = sum(pred_labels)\n    test_df.iloc[i, 1] = pred","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:04:34.598025Z","iopub.execute_input":"2022-04-02T22:04:34.598281Z","iopub.status.idle":"2022-04-02T22:04:47.072200Z","shell.execute_reply.started":"2022-04-02T22:04:34.598250Z","shell.execute_reply":"2022-04-02T22:04:47.071428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T22:06:32.170090Z","iopub.execute_input":"2022-04-02T22:06:32.170947Z","iopub.status.idle":"2022-04-02T22:06:32.221309Z","shell.execute_reply.started":"2022-04-02T22:06:32.170896Z","shell.execute_reply":"2022-04-02T22:06:32.220622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# P.S.\n\n","metadata":{}},{"cell_type":"markdown","source":"With this model I got the accuracy about 40% on the train set (about 80% accuracy per digit in average). \n\nUnfortunately, I don't have much time to improve this solution. \n\nBut there is many ways to improve results:\n* At first, really the most important possible improvement is changing of my generated dataset (I spend not so much time to create it, and I notice that it is not like original UltraMNIST). So, my models have a really good results with generated dataset, but on real UltraMNIST quality is significantly decreasing.\n* Second, you can tune many hyperparameters across my model (both for loss function and for non-max supression)\n* Finally, I think it may be useful to use even more YOLO models with different grid sizes.\n\nOf course, this solution is not an innovation. But I am new to CV, and this is my first serious work with object detection. \n\nI hope that this notebook will be useful for you. Thanks!","metadata":{}}]}