{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Google AI Open Images - Object Detection Track","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport torch \nimport cv2\nfrom PIL import Image, ImageColor, ImageDraw, ImageFont\nimport cv2, os, glob, time\nimport xml.etree.ElementTree as ET\nimport tensorflow as tf\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.layers import (\n    Add, Concatenate, Conv2D,\n    Input, Lambda, LeakyReLU,\n    MaxPool2D, UpSampling2D, ZeroPadding2D\n)\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.losses import (\n    binary_crossentropy,\n    sparse_categorical_crossentropy\n)\nfrom tensorflow.keras.utils import plot_model\n\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.702598Z","iopub.execute_input":"2022-10-01T12:33:30.703557Z","iopub.status.idle":"2022-10-01T12:33:30.711990Z","shell.execute_reply.started":"2022-10-01T12:33:30.703505Z","shell.execute_reply":"2022-10-01T12:33:30.710909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../input\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.718234Z","iopub.execute_input":"2022-10-01T12:33:30.718525Z","iopub.status.idle":"2022-10-01T12:33:30.730148Z","shell.execute_reply.started":"2022-10-01T12:33:30.718497Z","shell.execute_reply":"2022-10-01T12:33:30.729389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../input/google-ai-open-images-object-detection-track\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.731696Z","iopub.execute_input":"2022-10-01T12:33:30.732189Z","iopub.status.idle":"2022-10-01T12:33:30.743023Z","shell.execute_reply.started":"2022-10-01T12:33:30.732161Z","shell.execute_reply":"2022-10-01T12:33:30.741942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIR_PATH = '../input/google-ai-open-images-object-detection-track/'","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.744985Z","iopub.execute_input":"2022-10-01T12:33:30.745663Z","iopub.status.idle":"2022-10-01T12:33:30.751138Z","shell.execute_reply.started":"2022-10-01T12:33:30.745624Z","shell.execute_reply":"2022-10-01T12:33:30.750348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_image_paths = list(map(lambda x: DIR_PATH+'test/'+x, os.listdir(DIR_PATH+'test/')))\nall_image_ids   = list(path.split(DIR_PATH+'test/')[1][:-4] for path in all_image_paths)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.752319Z","iopub.execute_input":"2022-10-01T12:33:30.752695Z","iopub.status.idle":"2022-10-01T12:33:30.923413Z","shell.execute_reply.started":"2022-10-01T12:33:30.752655Z","shell.execute_reply":"2022-10-01T12:33:30.922318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('len(all_image_paths):', len(all_image_paths))\nprint('len(all_image_ids)  :', len(all_image_ids))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.925965Z","iopub.execute_input":"2022-10-01T12:33:30.926977Z","iopub.status.idle":"2022-10-01T12:33:30.932801Z","shell.execute_reply.started":"2022-10-01T12:33:30.926935Z","shell.execute_reply":"2022-10-01T12:33:30.931854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\n'.join(all_image_paths[:5]))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.934376Z","iopub.execute_input":"2022-10-01T12:33:30.934936Z","iopub.status.idle":"2022-10-01T12:33:30.943017Z","shell.execute_reply.started":"2022-10-01T12:33:30.934906Z","shell.execute_reply":"2022-10-01T12:33:30.942036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\n'.join(all_image_ids[:5]))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.943958Z","iopub.execute_input":"2022-10-01T12:33:30.944860Z","iopub.status.idle":"2022-10-01T12:33:30.952627Z","shell.execute_reply.started":"2022-10-01T12:33:30.944830Z","shell.execute_reply":"2022-10-01T12:33:30.951652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(i):\n    plt.imshow(Image.open(all_image_paths[i]))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.954698Z","iopub.execute_input":"2022-10-01T12:33:30.954987Z","iopub.status.idle":"2022-10-01T12:33:30.964322Z","shell.execute_reply.started":"2022-10-01T12:33:30.954962Z","shell.execute_reply":"2022-10-01T12:33:30.963429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The following GPU devices are available: %s\" % tf.test.gpu_device_name())","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.967300Z","iopub.execute_input":"2022-10-01T12:33:30.967543Z","iopub.status.idle":"2022-10-01T12:33:30.976449Z","shell.execute_reply.started":"2022-10-01T12:33:30.967520Z","shell.execute_reply":"2022-10-01T12:33:30.975371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.version","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.977825Z","iopub.execute_input":"2022-10-01T12:33:30.978150Z","iopub.status.idle":"2022-10-01T12:33:30.994694Z","shell.execute_reply.started":"2022-10-01T12:33:30.978121Z","shell.execute_reply":"2022-10-01T12:33:30.993941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"YOLOV3_LAYER_LIST = [\n    'yolo_darknet',\n    'yolo_conv_0',\n    'yolo_output_0',\n    'yolo_conv_1',\n    'yolo_output_1',\n    'yolo_conv_2',\n    'yolo_output_2',\n]\n\nYOLOV3_TINY_LAYER_LIST = [\n    'yolo_darknet',\n    'yolo_conv_0',\n    'yolo_output_0',\n    'yolo_conv_1',\n    'yolo_output_1',\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:30.995878Z","iopub.execute_input":"2022-10-01T12:33:30.996182Z","iopub.status.idle":"2022-10-01T12:33:31.004853Z","shell.execute_reply.started":"2022-10-01T12:33:30.996152Z","shell.execute_reply":"2022-10-01T12:33:31.003822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo_anchors = np.array([\n    (10, 13), (16, 30), (33, 23), (30, 61), (62, 45),\n    (59, 119), (116, 90), (156, 198), (373, 326)],\n    np.float32) / 416\n\nyolo_anchor_masks = np.array([[6, 7, 8], [3, 4, 5], [0, 1, 2]])\n\nyolo_tiny_anchors = np.array([\n    (10, 14), (23, 27), (37, 58),\n    (81, 82), (135, 169), (344, 319)],\n    np.float32) / 416\n\nyolo_tiny_anchor_masks = np.array([[3, 4, 5], [0, 1, 2]])","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.006008Z","iopub.execute_input":"2022-10-01T12:33:31.006304Z","iopub.status.idle":"2022-10-01T12:33:31.016172Z","shell.execute_reply.started":"2022-10-01T12:33:31.006276Z","shell.execute_reply":"2022-10-01T12:33:31.015176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = [\n    'person', 'bicycle','car','motorbike','aeroplane','bus','train','truck','boat',\n    'traffic light','fire hydrant','stop sign','parking meter','bench',\n    'bird','cat','dog','horse','sheep','cow','elephant','bear','zebra',\n    'giraffe','backpack','umbrella','handbag','tie','suitcase','frisbee',\n    'skis','snowboard','sports ball','kite','baseball bat','baseball glove',\n    'skateboard','surfboard','tennis racket','bottle','wine glass','cup',\n    'fork','knife','spoon','bowl','banana','apple','sandwich','orange',\n    'broccoli','carrot','hot dog','pizza','donut','cake','chair','sofa',\n    'pottedplant','bed','diningtable','toilet','tvmonitor','laptop','mouse',\n    'remote','keyboard','cell phone','microwave','oven','toaster','sink',\n    'refrigerator','book','clock','vase','scissors','teddy bear',\n    'hair drier','toothbrush'\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.017403Z","iopub.execute_input":"2022-10-01T12:33:31.017699Z","iopub.status.idle":"2022-10-01T12:33:31.032238Z","shell.execute_reply.started":"2022-10-01T12:33:31.017672Z","shell.execute_reply":"2022-10-01T12:33:31.031046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_darknet_weights(model, weights_file, tiny = False):\n    wf = open(weights_file, 'rb')\n    major, minor, revision, seen, _ = np.fromfile(wf, dtype=np.int32, count=5)\n    layers = YOLOV3_TINY_LAYER_LIST if tiny else YOLOV3_LAYER_LIST\n    for layer_name in layers:\n        sub_model = model.get_layer(layer_name)\n        for i, layer in enumerate(sub_model.layers):\n            if not layer.name.startswith('conv2d'):\n                continue\n            batch_norm = None\n            if i+1 < len(sub_model.layers) and \\\n               sub_model.layers[i+1].name.startswith('batch_norm'):\n                batch_norm = sub_model.layers[i+1]\n            filters = layer.filters\n            size = layer.kernel_size[0]\n            in_dim = layer.input_shape[-1]\n            if batch_norm is None:\n                conv_bias = np.fromfile(wf, dtype=np.float32, count=filters)\n            else:\n                bn_weights = np.fromfile(wf, dtype=np.float32, count=4 * filters). \\\n                             reshape((4, filters))[[1, 0, 2, 3]]\n            conv_shape = (filters, in_dim, size, size)\n            conv_weights = np.fromfile(wf, dtype=np.float32, count=np.product(conv_shape)). \\\n                           reshape(conv_shape).transpose([2, 3, 1, 0])\n            if batch_norm is None:\n                layer.set_weights([conv_weights, conv_bias])\n            else:\n                layer.set_weights([conv_weights])\n                batch_norm.set_weights(bn_weights)\n    wf.close()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.033626Z","iopub.execute_input":"2022-10-01T12:33:31.033984Z","iopub.status.idle":"2022-10-01T12:33:31.046221Z","shell.execute_reply.started":"2022-10-01T12:33:31.033953Z","shell.execute_reply":"2022-10-01T12:33:31.044901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def broadcast_iou(box_1, box_2):\n    box_1 = tf.expand_dims(box_1, -2)\n    box_2 = tf.expand_dims(box_2, 0)\n    new_shape = tf.broadcast_dynamic_shape(tf.shape(box_1), tf.shape(box_2))\n    box_1 = tf.broadcast_to(box_1, new_shape)\n    box_2 = tf.broadcast_to(box_2, new_shape)\n    int_w = tf.maximum(tf.minimum(box_1[..., 2], box_2[..., 2]) - \\\n            tf.maximum(box_1[..., 0], box_2[..., 0]), 0)\n    int_h = tf.maximum(tf.minimum(box_1[..., 3], box_2[..., 3]) - \\\n            tf.maximum(box_1[..., 1], box_2[..., 1]), 0)\n    int_area = int_w * int_h\n    box_1_area = (box_1[..., 2] - box_1[..., 0]) * (box_1[..., 3] - box_1[..., 1])\n    box_2_area = (box_2[..., 2] - box_2[..., 0]) * (box_2[..., 3] - box_2[..., 1])\n    iou =  int_area / (box_1_area + box_2_area - int_area)\n    return iou","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.047082Z","iopub.execute_input":"2022-10-01T12:33:31.047374Z","iopub.status.idle":"2022-10-01T12:33:31.064451Z","shell.execute_reply.started":"2022-10-01T12:33:31.047347Z","shell.execute_reply":"2022-10-01T12:33:31.063378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def freeze_all(model, frozen = True):\n    model.trainable = not frozen\n    if isinstance(model, tf.keras.Model):\n        for l in model.layers:\n            freeze_all(l, frozen)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.065771Z","iopub.execute_input":"2022-10-01T12:33:31.066117Z","iopub.status.idle":"2022-10-01T12:33:31.079303Z","shell.execute_reply.started":"2022-10-01T12:33:31.066084Z","shell.execute_reply":"2022-10-01T12:33:31.078269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_outputs(img, outputs, class_names):\n    boxes, objectness, classes, nums = outputs\n    boxes, objectness, classes, nums = boxes[0], objectness[0], classes[0], nums[0]\n    wh = np.flip(img.shape[0:2])\n    for i in range(nums):\n        x1y1 = tuple((np.array(boxes[i][0:2]) * wh).astype(np.int32))\n        x2y2 = tuple((np.array(boxes[i][2:4]) * wh).astype(np.int32))\n        img = cv2.rectangle(img, x1y1, x2y2, (255, 0, 0), 2)\n        img = cv2.putText(img, '{} {:.4f}'.format(class_names[int(classes[i])], objectness[i]),\n                          x1y1, cv2.FONT_HERSHEY_COMPLEX_SMALL, 1, (0, 0, 255), 2)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.080801Z","iopub.execute_input":"2022-10-01T12:33:31.081857Z","iopub.status.idle":"2022-10-01T12:33:31.090527Z","shell.execute_reply.started":"2022-10-01T12:33:31.081824Z","shell.execute_reply":"2022-10-01T12:33:31.089653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_labels(x, y, class_names):\n    img = x.numpy()\n    boxes, classes = tf.split(y, (4, 1), axis = -1)\n    classes = classes[..., 0]\n    wh = np.flip(img.shape[0 : 2])\n    for i in range(len(boxes)):\n        x1y1 = tuple((np.array(boxes[i][0:2]) * wh).astype(np.int32))\n        x2y2 = tuple((np.array(boxes[i][2:4]) * wh).astype(np.int32))\n        img = cv2.rectangle(img, x1y1, x2y2, (255, 0, 0), 2)\n        img = cv2.putText(img, class_names[classes[i]],\n                          x1y1, cv2.FONT_HERSHEY_COMPLEX_SMALL, 8, (0, 0, 255), 2)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.091806Z","iopub.execute_input":"2022-10-01T12:33:31.092710Z","iopub.status.idle":"2022-10-01T12:33:31.103447Z","shell.execute_reply.started":"2022-10-01T12:33:31.092666Z","shell.execute_reply":"2022-10-01T12:33:31.102618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform_images(x_train, size):\n    x_train = tf.image.resize(x_train, (size, size))\n    x_train = x_train / 255\n    return x_train","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.104597Z","iopub.execute_input":"2022-10-01T12:33:31.104840Z","iopub.status.idle":"2022-10-01T12:33:31.116727Z","shell.execute_reply.started":"2022-10-01T12:33:31.104817Z","shell.execute_reply":"2022-10-01T12:33:31.116027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tf.function\ndef transform_targets_for_output(y_true, grid_size, anchor_idxs, classes):\n    N = tf.shape(y_true)[0]\n    y_true_out  = tf.zeros((N, grid_size, grid_size, tf.shape(anchor_idxs)[0], 6))\n    anchor_idxs = tf.cast(anchor_idxs, tf.int32)\n    indexes = tf.TensorArray(tf.int32,   1, dynamic_size=True)\n    updates = tf.TensorArray(tf.float32, 1, dynamic_size=True)\n    idx = 0\n    for i in tf.range(N):\n        for j in tf.range(tf.shape(y_true)[1]):\n            if tf.equal(y_true[i][j][2], 0):\n                continue\n            anchor_eq = tf.equal(anchor_idxs, tf.cast(y_true[i][j][5], tf.int32))\n            if not tf.reduce_any(anchor_eq):\n                continue\n            box = y_true[i][j][0:4]\n            box_xy = (y_true[i][j][0:2] + y_true[i][j][2:4]) / 2\n            anchor_idx = tf.cast(tf.where(anchor_eq), tf.int32)\n            grid_xy = tf.cast(box_xy // (1/grid_size), tf.int32)\n            indexes = indexes.write(idx, [i, grid_xy[1], grid_xy[0], anchor_idx[0][0]])\n            updates = updates.write(idx, [box[0], box[1], box[2], box[3], 1, y_true[i][j][4]])\n            idx += 1\n    return tf.tensor_scatter_nd_update(y_true_out, indexes.stack(), updates.stack())","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.118356Z","iopub.execute_input":"2022-10-01T12:33:31.119093Z","iopub.status.idle":"2022-10-01T12:33:31.133040Z","shell.execute_reply.started":"2022-10-01T12:33:31.119038Z","shell.execute_reply":"2022-10-01T12:33:31.132042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform_targets(y_train, anchors, anchor_masks, classes):\n    y_outs = []\n    grid_size = 13\n    anchors = tf.cast(anchors, tf.float32)\n    anchor_area = anchors[..., 0] * anchors[..., 1]\n    box_wh = y_train[..., 2:4] - y_train[..., 0:2]\n    box_wh = tf.tile(tf.expand_dims(box_wh, -2), (1, 1, tf.shape(anchors)[0], 1))\n    box_area = box_wh[..., 0] * box_wh[..., 1]\n    intersection = tf.minimum(box_wh[..., 0], anchors[..., 0]) * \\\n                   tf.minimum(box_wh[..., 1], anchors[..., 1])\n    iou = intersection / (box_area + anchor_area - intersection)\n    anchor_idx = tf.cast(tf.argmax(iou, axis=-1), tf.float32)\n    anchor_idx = tf.expand_dims(anchor_idx, axis=-1)\n    y_train = tf.concat([y_train, anchor_idx], axis=-1)\n    for anchor_idxs in anchor_masks:\n        y_out = transform_targets_for_output(y_train, grid_size, anchor_idxs, classes)\n        y_outs.append(y_out)\n        grid_size *= 2\n    return tuple(y_outs)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.137964Z","iopub.execute_input":"2022-10-01T12:33:31.138565Z","iopub.status.idle":"2022-10-01T12:33:31.147199Z","shell.execute_reply.started":"2022-10-01T12:33:31.138532Z","shell.execute_reply":"2022-10-01T12:33:31.146408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BatchNormalization(tf.keras.layers.BatchNormalization):\n    def call(self, x, training = False):\n        if training is None:\n            traininig = tf.constant(False)\n        training = tf.logical_and(training, self.trainable)\n        return super().call(x, training)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.148562Z","iopub.execute_input":"2022-10-01T12:33:31.148847Z","iopub.status.idle":"2022-10-01T12:33:31.161506Z","shell.execute_reply.started":"2022-10-01T12:33:31.148820Z","shell.execute_reply":"2022-10-01T12:33:31.160734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def DarknetConv(x, filters, kernel_size, strides=1, batch_norm=True):\n    if strides == 1:\n        padding = 'same'\n    else:\n        padding = 'valid'\n        x = ZeroPadding2D(((1, 0), (1, 0)))(x)\n    x = Conv2D(filters=filters, \n               kernel_size=kernel_size,\n               strides=strides, \n               padding=padding,\n               use_bias=not batch_norm, \n               kernel_regularizer=l2(0.0005))(x)\n    if batch_norm:\n        x = BatchNormalization()(x)\n        x = LeakyReLU(alpha=0.1)(x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.162848Z","iopub.execute_input":"2022-10-01T12:33:31.163190Z","iopub.status.idle":"2022-10-01T12:33:31.171394Z","shell.execute_reply.started":"2022-10-01T12:33:31.163159Z","shell.execute_reply":"2022-10-01T12:33:31.170541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def DarknetResidual(x, filters):\n    prev_x = x\n    x = DarknetConv(x, filters // 2, 1)\n    x = DarknetConv(x, filters, 3)\n    x = Add()([prev_x, x])\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.172705Z","iopub.execute_input":"2022-10-01T12:33:31.173137Z","iopub.status.idle":"2022-10-01T12:33:31.181115Z","shell.execute_reply.started":"2022-10-01T12:33:31.173088Z","shell.execute_reply":"2022-10-01T12:33:31.180382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def DarknetBlock(x, filters, blocks):\n    x = DarknetConv(x, filters, 3, strides=2)\n    for _ in range(blocks):\n        x = DarknetResidual(x, filters)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.182226Z","iopub.execute_input":"2022-10-01T12:33:31.182687Z","iopub.status.idle":"2022-10-01T12:33:31.191114Z","shell.execute_reply.started":"2022-10-01T12:33:31.182659Z","shell.execute_reply":"2022-10-01T12:33:31.190283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Darknet(name=None):\n    x = inputs = Input([None, None, 3])\n    x = DarknetConv(x, 32, 3)\n    x = DarknetBlock(x, 64, 1)\n    x = DarknetBlock(x, 128, 2)\n    x = x_36 = DarknetBlock(x, 256, 8)\n    x = x_61 = DarknetBlock(x, 512, 8)\n    x = DarknetBlock(x, 1024, 4)\n    return tf.keras.Model(inputs, (x_36, x_61, x), name=name)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.192388Z","iopub.execute_input":"2022-10-01T12:33:31.192937Z","iopub.status.idle":"2022-10-01T12:33:31.208538Z","shell.execute_reply.started":"2022-10-01T12:33:31.192895Z","shell.execute_reply":"2022-10-01T12:33:31.207803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def DarknetTiny(name=None):\n    x = inputs = Input([None, None, 3])\n    x = DarknetConv(x, 16, 3)\n    x = MaxPool2D(2, 2, 'same')(x)\n    x = DarknetConv(x, 32, 3)\n    x = MaxPool2D(2, 2, 'same')(x)\n    x = DarknetConv(x, 64, 3)\n    x = MaxPool2D(2, 2, 'same')(x)\n    x = DarknetConv(x, 128, 3)\n    x = MaxPool2D(2, 2, 'same')(x)\n    x = x_8 = DarknetConv(x, 256, 3)  # skip connection\n    x = MaxPool2D(2, 2, 'same')(x)\n    x = DarknetConv(x, 512, 3)\n    x = MaxPool2D(2, 1, 'same')(x)\n    x = DarknetConv(x, 1024, 3)\n    return tf.keras.Model(inputs, (x_8, x), name=name)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.209448Z","iopub.execute_input":"2022-10-01T12:33:31.210124Z","iopub.status.idle":"2022-10-01T12:33:31.221401Z","shell.execute_reply.started":"2022-10-01T12:33:31.210079Z","shell.execute_reply":"2022-10-01T12:33:31.220657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloConv(filters, name=None):\n    def yolo_conv(x_in):\n        if isinstance(x_in, tuple):\n            inputs = Input(x_in[0].shape[1:]), Input(x_in[1].shape[1:])\n            x, x_skip = inputs\n            # concat with skip connection\n            x = DarknetConv(x, filters, 1)\n            x = UpSampling2D(2)(x)\n            x = Concatenate()([x, x_skip])\n        else:\n            x = inputs = Input(x_in.shape[1:])\n        x = DarknetConv(x, filters, 1)\n        x = DarknetConv(x, filters * 2, 3)\n        x = DarknetConv(x, filters, 1)\n        x = DarknetConv(x, filters * 2, 3)\n        x = DarknetConv(x, filters, 1)\n        return Model(inputs, x, name=name)(x_in)\n    return yolo_conv","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.222270Z","iopub.execute_input":"2022-10-01T12:33:31.222773Z","iopub.status.idle":"2022-10-01T12:33:31.234487Z","shell.execute_reply.started":"2022-10-01T12:33:31.222746Z","shell.execute_reply":"2022-10-01T12:33:31.233496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloConvTiny(filters, name=None):\n    def yolo_conv(x_in):\n        if isinstance(x_in, tuple):\n            inputs = Input(x_in[0].shape[1:]), Input(x_in[1].shape[1:])\n            x, x_skip = inputs\n            # concat with skip connection\n            x = DarknetConv(x, filters, 1)\n            x = UpSampling2D(2)(x)\n            x = Concatenate()([x, x_skip])\n        else:\n            x = inputs = Input(x_in.shape[1:])\n            x = DarknetConv(x, filters, 1)\n        return Model(inputs, x, name=name)(x_in)\n    return yolo_conv","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.236165Z","iopub.execute_input":"2022-10-01T12:33:31.236538Z","iopub.status.idle":"2022-10-01T12:33:31.247604Z","shell.execute_reply.started":"2022-10-01T12:33:31.236500Z","shell.execute_reply":"2022-10-01T12:33:31.246853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloOutput(filters, anchors, classes, name=None):\n    def yolo_output(x_in):\n        x = inputs = Input(x_in.shape[1:])\n        x = DarknetConv(x, filters * 2, 3)\n        x = DarknetConv(x, anchors * (classes + 5), 1, batch_norm=False)\n        x = Lambda(lambda x: tf.reshape(x, (-1, tf.shape(x)[1], tf.shape(x)[2], anchors, classes + 5)))(x)\n        return tf.keras.Model(inputs, x, name=name)(x_in)\n    return yolo_output","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.248850Z","iopub.execute_input":"2022-10-01T12:33:31.249488Z","iopub.status.idle":"2022-10-01T12:33:31.263875Z","shell.execute_reply.started":"2022-10-01T12:33:31.249457Z","shell.execute_reply":"2022-10-01T12:33:31.263204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo_boxes(pred, anchors, classes):\n    grid_size = tf.shape(pred)[1]\n    box_xy, box_wh, objectness, class_probs = tf.split(\n        pred, (2, 2, 1, classes), axis=-1)\n    box_xy = tf.sigmoid(box_xy)\n    objectness = tf.sigmoid(objectness)\n    class_probs = tf.sigmoid(class_probs)\n    pred_box = tf.concat((box_xy, box_wh), axis=-1)  # original xywh for loss\n    grid = tf.meshgrid(tf.range(grid_size), tf.range(grid_size))\n    grid = tf.expand_dims(tf.stack(grid, axis=-1), axis=2)  # [gx, gy, 1, 2]\n    box_xy = (box_xy + tf.cast(grid, tf.float32)) / \\\n        tf.cast(grid_size, tf.float32)\n    box_wh = tf.exp(box_wh) * anchors\n    box_x1y1 = box_xy - box_wh / 2\n    box_x2y2 = box_xy + box_wh / 2\n    bbox = tf.concat([box_x1y1, box_x2y2], axis=-1)\n    return bbox, objectness, class_probs, pred_box","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.265077Z","iopub.execute_input":"2022-10-01T12:33:31.265559Z","iopub.status.idle":"2022-10-01T12:33:31.275088Z","shell.execute_reply.started":"2022-10-01T12:33:31.265530Z","shell.execute_reply":"2022-10-01T12:33:31.274215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo_nms(outputs, anchors, masks, classes):\n    b, c, t = [], [], []\n    for o in outputs:\n        b.append(tf.reshape(o[0], (tf.shape(o[0])[0], -1, tf.shape(o[0])[-1])))\n        c.append(tf.reshape(o[1], (tf.shape(o[1])[0], -1, tf.shape(o[1])[-1])))\n        t.append(tf.reshape(o[2], (tf.shape(o[2])[0], -1, tf.shape(o[2])[-1])))\n    bbox = tf.concat(b, axis=1)\n    confidence = tf.concat(c, axis=1)\n    class_probs = tf.concat(t, axis=1)\n    scores = confidence * class_probs\n    boxes, scores, classes, valid_detections = tf.image.combined_non_max_suppression(\n        boxes=tf.reshape(bbox, (tf.shape(bbox)[0], -1, 1, 4)),\n        scores=tf.reshape(\n            scores,\n            (tf.shape(scores)[0], -1, tf.shape(scores)[-1])\n        ),\n        max_output_size_per_class=100,\n        max_total_size = 100,\n        iou_threshold = 0.5,\n        score_threshold = 0.5\n    )\n    return boxes, scores, classes, valid_detections","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.276380Z","iopub.execute_input":"2022-10-01T12:33:31.276702Z","iopub.status.idle":"2022-10-01T12:33:31.288753Z","shell.execute_reply.started":"2022-10-01T12:33:31.276670Z","shell.execute_reply":"2022-10-01T12:33:31.287957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloV3(size=None, channels=3, anchors=yolo_anchors, masks=yolo_anchor_masks, classes=80, training=False):\n    x = inputs = Input([size, size, channels])\n    x_36, x_61, x = Darknet(name='yolo_darknet')(x)\n    x = YoloConv(512, name='yolo_conv_0')(x)\n    output_0 = YoloOutput(512, len(masks[0]), classes, name='yolo_output_0')(x)\n    x = YoloConv(256, name='yolo_conv_1')((x, x_61))\n    output_1 = YoloOutput(256, len(masks[1]), classes, name='yolo_output_1')(x)\n    x = YoloConv(128, name='yolo_conv_2')((x, x_36))\n    output_2 = YoloOutput(128, len(masks[2]), classes, name='yolo_output_2')(x)\n    if training:\n        return Model(inputs, (output_0, output_1, output_2), name='yolov3')\n    boxes_0 = Lambda(lambda x: yolo_boxes(x, anchors[masks[0]], classes),\n                     name='yolo_boxes_0')(output_0)\n    boxes_1 = Lambda(lambda x: yolo_boxes(x, anchors[masks[1]], classes),\n                     name='yolo_boxes_1')(output_1)\n    boxes_2 = Lambda(lambda x: yolo_boxes(x, anchors[masks[2]], classes),\n                     name='yolo_boxes_2')(output_2)\n    outputs = Lambda(lambda x: yolo_nms(x, anchors, masks, classes),\n                     name='yolo_nms')((boxes_0[:3], boxes_1[:3], boxes_2[:3]))\n    return Model(inputs, outputs, name='yolov3')","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.289889Z","iopub.execute_input":"2022-10-01T12:33:31.290688Z","iopub.status.idle":"2022-10-01T12:33:31.304747Z","shell.execute_reply.started":"2022-10-01T12:33:31.290648Z","shell.execute_reply":"2022-10-01T12:33:31.303972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloV3Tiny(size=None, channels=3, anchors=yolo_tiny_anchors, masks=yolo_tiny_anchor_masks, classes=80, training=False):\n    x = inputs = Input([size, size, channels])\n    x_8, x = DarknetTiny(name='yolo_darknet')(x)\n    x = YoloConvTiny(256, name='yolo_conv_0')(x)\n    output_0 = YoloOutput(256, len(masks[0]), classes, name='yolo_output_0')(x)\n    x = YoloConvTiny(128, name='yolo_conv_1')((x, x_8))\n    output_1 = YoloOutput(128, len(masks[1]), classes, name='yolo_output_1')(x)\n    if training:\n        return Model(inputs, (output_0, output_1), name='yolov3')\n    boxes_0 = Lambda(lambda x: yolo_boxes(x, anchors[masks[0]], classes),\n                     name='yolo_boxes_0')(output_0)\n    boxes_1 = Lambda(lambda x: yolo_boxes(x, anchors[masks[1]], classes),\n                     name='yolo_boxes_1')(output_1)\n    outputs = Lambda(lambda x: yolo_nms(x, anchors, masks, classes),\n                     name='yolo_nms')((boxes_0[:3], boxes_1[:3]))\n    return Model(inputs, outputs, name='yolov3_tiny')","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.305889Z","iopub.execute_input":"2022-10-01T12:33:31.306787Z","iopub.status.idle":"2022-10-01T12:33:31.319039Z","shell.execute_reply.started":"2022-10-01T12:33:31.306745Z","shell.execute_reply":"2022-10-01T12:33:31.318326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def YoloLoss(anchors, classes=80, ignore_thresh=0.5):\n    def yolo_loss(y_true, y_pred):\n        # 1. transform all pred outputs\n        # y_pred: (batch_size, grid, grid, anchors, (x, y, w, h, obj, ...cls))\n        pred_box, pred_obj, pred_class, pred_xywh = yolo_boxes(y_pred, anchors, classes)\n        pred_xy = pred_xywh[..., 0:2]\n        pred_wh = pred_xywh[..., 2:4]\n        # 2. transform all true outputs\n        # y_true: (batch_size, grid, grid, anchors, (x1, y1, x2, y2, obj, cls))\n        true_box, true_obj, true_class_idx = tf.split(\n            y_true, (4, 1, 1), axis=-1)\n        true_xy = (true_box[..., 0:2] + true_box[..., 2:4]) / 2\n        true_wh = true_box[..., 2:4] - true_box[..., 0:2]\n        # give higher weights to small boxes\n        box_loss_scale = 2 - true_wh[..., 0] * true_wh[..., 1]\n        # 3. inverting the pred box equations\n        grid_size = tf.shape(y_true)[1]\n        grid = tf.meshgrid(tf.range(grid_size), tf.range(grid_size))\n        grid = tf.expand_dims(tf.stack(grid, axis=-1), axis=2)\n        true_xy = true_xy * tf.cast(grid_size, tf.float32) - \\\n            tf.cast(grid, tf.float32)\n        true_wh = tf.math.log(true_wh / anchors)\n        true_wh = tf.where(tf.math.is_inf(true_wh), tf.zeros_like(true_wh), true_wh)\n        # 4. calculate all masks\n        obj_mask = tf.squeeze(true_obj, -1)\n        # ignore false positive when iou is over threshold\n        true_box_flat = tf.boolean_mask(true_box, tf.cast(obj_mask, tf.bool))\n        best_iou = tf.reduce_max(broadcast_iou(\n            pred_box, true_box_flat), axis=-1)\n        ignore_mask = tf.cast(best_iou < ignore_thresh, tf.float32)\n        # 5. calculate all losses\n        xy_loss = obj_mask * box_loss_scale * \\\n            tf.reduce_sum(tf.square(true_xy - pred_xy), axis=-1)\n        wh_loss = obj_mask * box_loss_scale * \\\n            tf.reduce_sum(tf.square(true_wh - pred_wh), axis=-1)\n        obj_loss = binary_crossentropy(true_obj, pred_obj)\n        obj_loss = obj_mask * obj_loss + \\\n            (1 - obj_mask) * ignore_mask * obj_loss\n        # Could also use binary_crossentropy instead\n        class_loss = obj_mask * sparse_categorical_crossentropy(\n            true_class_idx, pred_class)\n        # 6. sum over (batch, gridx, gridy, anchors) => (batch, 1)\n        xy_loss = tf.reduce_sum(xy_loss, axis=(1, 2, 3))\n        wh_loss = tf.reduce_sum(wh_loss, axis=(1, 2, 3))\n        obj_loss = tf.reduce_sum(obj_loss, axis=(1, 2, 3))\n        class_loss = tf.reduce_sum(class_loss, axis=(1, 2, 3))\n        return xy_loss + wh_loss + obj_loss + class_loss\n    return yolo_loss","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.320268Z","iopub.execute_input":"2022-10-01T12:33:31.320796Z","iopub.status.idle":"2022-10-01T12:33:31.336554Z","shell.execute_reply.started":"2022-10-01T12:33:31.320768Z","shell.execute_reply":"2022-10-01T12:33:31.335674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo = YoloV3(classes = 80)\nyolo.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:31.337541Z","iopub.execute_input":"2022-10-01T12:33:31.337814Z","iopub.status.idle":"2022-10-01T12:33:33.943471Z","shell.execute_reply.started":"2022-10-01T12:33:31.337773Z","shell.execute_reply":"2022-10-01T12:33:33.942415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_model_arch():\n    plot_model(\n        yolo, rankdir = 'TB',\n        to_file = 'yolo_model.png',\n        show_shapes = False,\n        show_layer_names = True,\n        expand_nested = True\n    )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-01T12:33:33.949631Z","iopub.execute_input":"2022-10-01T12:33:33.950212Z","iopub.status.idle":"2022-10-01T12:33:33.955326Z","shell.execute_reply.started":"2022-10-01T12:33:33.950162Z","shell.execute_reply":"2022-10-01T12:33:33.954252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir data\n!wget wget https://pjreddie.com/media/files/yolov3.weights -O data/yolov3.weights\n!wget https://pjreddie.com/media/files/yolov3-tiny.weights -O data/yolov3-tiny.weights","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:33.957023Z","iopub.execute_input":"2022-10-01T12:33:33.957404Z","iopub.status.idle":"2022-10-01T12:33:40.461641Z","shell.execute_reply.started":"2022-10-01T12:33:33.957366Z","shell.execute_reply":"2022-10-01T12:33:40.460526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_darknet_weights(yolo, './data/yolov3.weights', False)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:40.463398Z","iopub.execute_input":"2022-10-01T12:33:40.463837Z","iopub.status.idle":"2022-10-01T12:33:41.027504Z","shell.execute_reply.started":"2022-10-01T12:33:40.463787Z","shell.execute_reply":"2022-10-01T12:33:41.026320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(image_file, visualize = True, figsize = (16, 16)):\n    img = tf.image.decode_image(open(image_file, 'rb').read(), channels=3)\n    img = tf.expand_dims(img, 0)\n    img = transform_images(img, 416)\n    boxes, scores, classes, nums = yolo.predict(img)\n    deleted_idx = [i for i in range(len(scores[0])) if scores[0][i] == 0]\n    boxes   = np.delete(boxes,   deleted_idx, axis=1)\n    scores  = np.delete(scores,  deleted_idx, axis=1)\n    classes = np.delete(classes, deleted_idx, axis=1)\n    img = cv2.cvtColor(cv2.imread(image_file), cv2.COLOR_BGR2RGB)\n    img = draw_outputs(img, (boxes, scores, classes, nums), class_names)\n    if visualize:\n        fig, axes = plt.subplots(figsize = figsize)\n        plt.imshow(img)\n        plt.show()\n    return boxes, scores, classes, nums","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:41.029417Z","iopub.execute_input":"2022-10-01T12:33:41.029863Z","iopub.status.idle":"2022-10-01T12:33:41.040191Z","shell.execute_reply.started":"2022-10-01T12:33:41.029817Z","shell.execute_reply":"2022-10-01T12:33:41.039056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_prediction_string(boxes, scores, classes, nums):\n    pred_strs = []\n    for i, score in enumerate(scores):\n        single_pred_str = \"\"\n        single_pred_str += str(classes[i]) + \" \" + str(score) + \" \"\n        single_pred_str += \" \".join(str(x) for x in boxes[i])\n        pred_strs.append(single_pred_str)\n    return ' '.join(map(str, pred_strs))","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:41.041745Z","iopub.execute_input":"2022-10-01T12:33:41.042074Z","iopub.status.idle":"2022-10-01T12:33:41.052617Z","shell.execute_reply.started":"2022-10-01T12:33:41.042044Z","shell.execute_reply":"2022-10-01T12:33:41.051954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes, scores, classes, nums = predict(all_image_paths[30], figsize = (10, 10))\nprint('boxes.shape  :', boxes.shape,   boxes)\nprint('scores.shape :', scores.shape,  scores)\nprint('classes.shape:', classes.shape, classes)\nprint('nums.shape   :', nums.shape,    nums)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:41.053779Z","iopub.execute_input":"2022-10-01T12:33:41.054311Z","iopub.status.idle":"2022-10-01T12:33:44.565268Z","shell.execute_reply.started":"2022-10-01T12:33:41.054280Z","shell.execute_reply":"2022-10-01T12:33:44.564266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_prediction_entry(i, boxes, scores, classes, nums):\n    return {\n        \"ImageID\": all_image_ids[i],\n        \"PredictionString\": get_prediction_string(boxes, scores, classes, nums)\n    }","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:44.566395Z","iopub.execute_input":"2022-10-01T12:33:44.566679Z","iopub.status.idle":"2022-10-01T12:33:44.572599Z","shell.execute_reply.started":"2022-10-01T12:33:44.566644Z","shell.execute_reply":"2022-10-01T12:33:44.571402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = [get_prediction_entry(i, boxes[0], scores[0], classes[0], nums) for i in range(99999)]","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:44.574173Z","iopub.execute_input":"2022-10-01T12:33:44.574500Z","iopub.status.idle":"2022-10-01T12:33:46.131052Z","shell.execute_reply.started":"2022-10-01T12:33:44.574472Z","shell.execute_reply":"2022-10-01T12:33:46.130183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df = pd.DataFrame(predictions)\npredictions_df","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:46.132426Z","iopub.execute_input":"2022-10-01T12:33:46.132804Z","iopub.status.idle":"2022-10-01T12:33:46.230151Z","shell.execute_reply.started":"2022-10-01T12:33:46.132767Z","shell.execute_reply":"2022-10-01T12:33:46.229343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(DIR_PATH+'sample_submission.csv')\nsubmission_df.update(predictions_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:46.231409Z","iopub.execute_input":"2022-10-01T12:33:46.231702Z","iopub.status.idle":"2022-10-01T12:33:46.376245Z","shell.execute_reply.started":"2022-10-01T12:33:46.231673Z","shell.execute_reply":"2022-10-01T12:33:46.375139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('./prediction.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T12:33:46.377719Z","iopub.execute_input":"2022-10-01T12:33:46.378126Z","iopub.status.idle":"2022-10-01T12:33:46.727256Z","shell.execute_reply.started":"2022-10-01T12:33:46.378091Z","shell.execute_reply":"2022-10-01T12:33:46.725855Z"},"trusted":true},"execution_count":null,"outputs":[]}]}