{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Необходимо для визуализации\n# Required for visualization\n!pip install pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:45:18.709988Z","iopub.execute_input":"2021-06-21T15:45:18.710433Z","iopub.status.idle":"2021-06-21T15:45:35.166891Z","shell.execute_reply.started":"2021-06-21T15:45:18.710346Z","shell.execute_reply":"2021-06-21T15:45:35.165907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Нужно для изменения размера bbox при изменении размера изображения\n# Needed to resize bbox when resizing image\n!pip install ../input/biblioteki/bibliot/albumentations-0.5.2","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:46:42.219311Z","iopub.execute_input":"2021-06-21T15:46:42.219694Z","iopub.status.idle":"2021-06-21T15:46:50.773170Z","shell.execute_reply.started":"2021-06-21T15:46:42.219652Z","shell.execute_reply":"2021-06-21T15:46:50.772220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\nimport numpy as np\nimport json\nimport ast\nimport pandas as pd\nimport datetime\nimport cv2\nimport albumentations as A","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:46:57.484074Z","iopub.execute_input":"2021-06-21T15:46:57.484430Z","iopub.status.idle":"2021-06-21T15:46:59.443476Z","shell.execute_reply.started":"2021-06-21T15:46:57.484395Z","shell.execute_reply":"2021-06-21T15:46:59.442711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Отбираем изображения, для которых имеются данные о координатах BBOX\n# Selecting images for which there is data on BBOX coordinates\ntr = pd.read_csv('../input/binary-csv/bincsv/train_binary.csv')\nvl = pd.read_csv('../input/binary-csv/bincsv/val_binary.csv')\nvl.loc[vl['id']=='2f12dbb2caf2']","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:05.941449Z","iopub.execute_input":"2021-06-21T15:47:05.942118Z","iopub.status.idle":"2021-06-21T15:47:06.055855Z","shell.execute_reply.started":"2021-06-21T15:47:05.942082Z","shell.execute_reply":"2021-06-21T15:47:06.054885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For training with 4 classes\ntr = tr[(tr['label_encoded']>0) & (tr['label']!='none 1 0 0 1 1')]\ntr","metadata":{"execution":{"iopub.status.busy":"2021-05-30T16:00:53.004717Z","iopub.execute_input":"2021-05-30T16:00:53.005436Z","iopub.status.idle":"2021-05-30T16:00:53.036864Z","shell.execute_reply.started":"2021-05-30T16:00:53.005383Z","shell.execute_reply":"2021-05-30T16:00:53.035641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For training with 2 classes\ntr = tr[(tr['binary_label']>0) & (tr['label']!='none 1 0 0 1 1')]\ntr","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:21.403771Z","iopub.execute_input":"2021-06-21T15:47:21.404176Z","iopub.status.idle":"2021-06-21T15:47:21.449144Z","shell.execute_reply.started":"2021-06-21T15:47:21.404141Z","shell.execute_reply":"2021-06-21T15:47:21.448120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr.to_csv('./tr_last.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:32.499634Z","iopub.execute_input":"2021-06-21T15:47:32.500042Z","iopub.status.idle":"2021-06-21T15:47:32.561323Z","shell.execute_reply.started":"2021-06-21T15:47:32.500005Z","shell.execute_reply":"2021-06-21T15:47:32.560179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tr)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:36.117412Z","iopub.execute_input":"2021-06-21T15:47:36.117755Z","iopub.status.idle":"2021-06-21T15:47:36.123549Z","shell.execute_reply.started":"2021-06-21T15:47:36.117724Z","shell.execute_reply":"2021-06-21T15:47:36.122352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For training with 4 classes\nvl = vl[(vl['label_encoded']>0) & (vl['label']!='none 1 0 0 1 1')]","metadata":{"execution":{"iopub.status.busy":"2021-05-30T16:01:45.311636Z","iopub.execute_input":"2021-05-30T16:01:45.312384Z","iopub.status.idle":"2021-05-30T16:01:45.321892Z","shell.execute_reply.started":"2021-05-30T16:01:45.31233Z","shell.execute_reply":"2021-05-30T16:01:45.320801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For training with 2 classes\nvl = vl[(vl['binary_label']>0) & (vl['label']!='none 1 0 0 1 1')]","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:41.147132Z","iopub.execute_input":"2021-06-21T15:47:41.147490Z","iopub.status.idle":"2021-06-21T15:47:41.154596Z","shell.execute_reply.started":"2021-06-21T15:47:41.147457Z","shell.execute_reply":"2021-06-21T15:47:41.153473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vl.to_csv('./vl_last.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:45.219888Z","iopub.execute_input":"2021-06-21T15:47:45.220432Z","iopub.status.idle":"2021-06-21T15:47:45.239743Z","shell.execute_reply.started":"2021-06-21T15:47:45.220385Z","shell.execute_reply":"2021-06-21T15:47:45.238697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vl)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:46.978719Z","iopub.execute_input":"2021-06-21T15:47:46.979144Z","iopub.status.idle":"2021-06-21T15:47:46.984884Z","shell.execute_reply.started":"2021-06-21T15:47:46.979103Z","shell.execute_reply":"2021-06-21T15:47:46.983933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = pd.read_csv('./tr_last.csv')\nvl = pd.read_csv('./vl_last.csv')\nvl.loc[vl['id']=='2f12dbb2caf2']","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:51.418333Z","iopub.execute_input":"2021-06-21T15:47:51.418904Z","iopub.status.idle":"2021-06-21T15:47:51.465292Z","shell.execute_reply.started":"2021-06-21T15:47:51.418836Z","shell.execute_reply":"2021-06-21T15:47:51.464088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Labels for classification\n# Метки для классификации\nlabels = {\n0:\"Negative for Pneumonia\",\n1:\"Typical Appearance\",\n2:\"Indeterminate Appearance\",\n3:\"Atypical Appearance\"}","metadata":{"execution":{"iopub.status.busy":"2021-05-30T16:02:08.405661Z","iopub.execute_input":"2021-05-30T16:02:08.406058Z","iopub.status.idle":"2021-05-30T16:02:08.410652Z","shell.execute_reply.started":"2021-05-30T16:02:08.406002Z","shell.execute_reply":"2021-05-30T16:02:08.409884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Labels for classification\n# Метки для классификации\nlabels = {\n0:\"negative\",\n1:\"opacity\"}","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:56.604944Z","iopub.execute_input":"2021-06-21T15:47:56.605339Z","iopub.status.idle":"2021-06-21T15:47:56.610290Z","shell.execute_reply.started":"2021-06-21T15:47:56.605302Z","shell.execute_reply":"2021-06-21T15:47:56.609115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"now = datetime.datetime.now()\n\ndata = dict(\n    images=[\n        # license, url, file_name, height, width, date_captured, id\n    ],\n    type='instances',\n    annotations=[\n        # segmentation, area, iscrowd, image_id, bbox, category_id, id\n    ],\n    categories=[\n        # supercategory, id, name\n    ],\n)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:47:59.724930Z","iopub.execute_input":"2021-06-21T15:47:59.725298Z","iopub.status.idle":"2021-06-21T15:47:59.730455Z","shell.execute_reply.started":"2021-06-21T15:47:59.725268Z","shell.execute_reply":"2021-06-21T15:47:59.729140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_name_to_id = {}\nfor i, each_label in enumerate(labels):\n    class_name = each_label\n    class_id = i\n    class_name_to_id[class_name] = class_id\n    data['categories'].append(dict(\n        supercategory=None,\n        id=class_id,\n        name=str(class_name),\n    ))","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:48:05.397396Z","iopub.execute_input":"2021-06-21T15:48:05.397771Z","iopub.status.idle":"2021-06-21T15:48:05.403623Z","shell.execute_reply.started":"2021-06-21T15:48:05.397739Z","shell.execute_reply":"2021-06-21T15:48:05.402399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:48:07.876409Z","iopub.execute_input":"2021-06-21T15:48:07.876775Z","iopub.status.idle":"2021-06-21T15:48:07.882855Z","shell.execute_reply.started":"2021-06-21T15:48:07.876743Z","shell.execute_reply":"2021-06-21T15:48:07.881935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Определены следующие функции API:\n# encode - Кодировать двоичные маски с помощью RLE.\n# decode - декодировать двоичные маски, закодированные с помощью RLE.\n# merge - вычислить объединение или пересечение закодированных масок.\n# iou - вычислить пересечение по объединению масок.\n# area - Расчетная область закодированных масок.\n# toBbox - получить ограничивающие рамки, окружающие закодированные маски.\n# frPyObjects - Преобразование многоугольника, bbox и несжатого RLE в закодированную маску RLE.\n\n#  Rs     = encode( masks )\n#  masks  = decode( Rs )\n#  R      = merge( Rs, intersect=false )\n#  o      = iou( dt, gt, iscrowd )\n#  a      = area( Rs )\n#  bbs    = toBbox( Rs )\n#  Rs     = frPyObjects( [pyObjects], h, w )\n\n# In the API the following formats are used:\n#  Rs      - [dict] Run-length encoding of binary masks\n#  R       - dict Run-length encoding of binary mask\n#  masks   - [hxwxn] Binary mask(s) (must have type np.ndarray(dtype=uint8) in column-major order)\n#  iscrowd - [nx1] list of np.ndarray. 1 indicates corresponding gt image has crowd region to ignore\n#  bbs     - [nx4] Bounding box(es) stored as [x y w h]\n#  poly    - Polygon stored as [[x1 y1 x2 y2...],[x1 y1 ...],...] (2D list)\n#  dt,gt   - May be either bounding boxes or encoded masks\n# Both poly and bbs are 0-indexed (bbox=[0 0 1 1] encloses first pixel).","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# JSON для изображений 512*512\n# JSON for 512 * 512 images\ntrain_out_file = './train_1280_annotations.json'\nval_out_file = './val_1280_annotations.json'\ndata_val = data.copy()\ndata_val['images'] = []\ndata_val['annotations'] = []\ndata_train = data.copy()\ndata_train['images'] = []\ndata_train['annotations'] = []\n# JSON для оригинальных изображений\n# JSON original images\ntrain_original_out_file = './train_annotations.json'\nval_original_out_file = './val_annotations.json'\ndata_original_val = data.copy()\ndata_original_val['images'] = []\ndata_original_val['annotations'] = []\ndata_original_train = data.copy()\ndata_original_train['images'] = []\ndata_original_train['annotations'] = []","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:48:42.189144Z","iopub.execute_input":"2021-06-21T15:48:42.189506Z","iopub.status.idle":"2021-06-21T15:48:42.196361Z","shell.execute_reply.started":"2021-06-21T15:48:42.189473Z","shell.execute_reply":"2021-06-21T15:48:42.195293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# JSON для измененных изображений\n# JSON for modified images","metadata":{}},{"cell_type":"code","source":"len(tr)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:48:55.301847Z","iopub.execute_input":"2021-06-21T15:48:55.302330Z","iopub.status.idle":"2021-06-21T15:48:55.308904Z","shell.execute_reply.started":"2021-06-21T15:48:55.302283Z","shell.execute_reply":"2021-06-21T15:48:55.307781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im_size = 512","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Трансформатор изображения и bbox\n# Image transformer and bbox\ntransform = A.Compose(\n    [\n        A.Resize(height = im_size , width = im_size, p=1),\n    ], \n    p=1.0,  bbox_params=A.BboxParams( format='coco', min_area=0,  min_visibility=0, label_fields=['labels']  ))        ","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:50:22.766657Z","iopub.execute_input":"2021-06-21T15:50:22.767043Z","iopub.status.idle":"2021-06-21T15:50:22.771483Z","shell.execute_reply.started":"2021-06-21T15:50:22.767011Z","shell.execute_reply":"2021-06-21T15:50:22.770836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in range(len(tr)):\n    image_id = tr.iloc[idx].id\n    class_id = tr.iloc[idx].label_encoded #binary_label\n    path = tr.iloc[idx].path\n    img = cv2.imread((path), cv2.IMREAD_COLOR)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)#.astype(np.float32)\n    width = im_size #int(tr.iloc[idx].w)\n    height = im_size #int(tr.iloc[idx].h)\n    data_train['images'].append(dict(\n            file_name = image_id+'.png',\n            width = width,\n            height = height,\n            date_captured=None,\n            id=int(idx)))\n\n    try:\n        a = tr.iloc[idx].boxes\n        boxes = ast.literal_eval(a)\n        for box in boxes:\n            x1= float(box['x'])\n            y1 = float(box['y'])\n            w = float(box['width'])\n            h = float(box['height'])\n            bb = []\n            bbox = [\n                    int(x1),\n                    int(y1),\n                    int(w),\n                    int(h)]\n            bb.append(bbox)\n            a = int(class_id)\n            b = str(class_id)\n            ct = {f'{a}:{b}'}\n            sample = transform(image=img, bboxes=bb, labels=b)\n            tr_boxes = sample['bboxes']\n            x_1 = tr_boxes[0][0]\n            y_1 = tr_boxes[0][1]\n            w_1 = tr_boxes[0][2]\n            h_1 = tr_boxes[0][3]\n            b_b = [int(x_1), int(y_1), int(w_1), int(h_1)]\n            tr_labels = sample['labels']\n            area = (w_1)*(h_1)\n            data_train['annotations'].append(dict(id=len(data_train['annotations']),\n                                                          area=round(area,3), \n                                                          bbox=b_b,\n                                                          iscrowd= 0, #1,\n                                                          image_id=int(idx),\n                                                          category_id=int(tr_labels[0])))\n                                                          #segmentation = rl ))\n    except ValueError:\n        print(image_id)\n    \n#     with open(train_out_file, 'w') as f:\n#         json.dump(data_train, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:50:24.917057Z","iopub.execute_input":"2021-06-21T15:50:24.917586Z","iopub.status.idle":"2021-06-21T15:56:16.596328Z","shell.execute_reply.started":"2021-06-21T15:50:24.917554Z","shell.execute_reply":"2021-06-21T15:56:16.595206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def myconverter(obj):\n    if isinstance(obj, np.integer):\n        return int(obj)\n    elif isinstance(obj, np.floating):\n        return float(obj)\n    elif isinstance(obj, np.ndarray):\n        return obj.tolist()\n    elif isinstance(obj, datetime.datetime):\n        return obj.__str__()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:57:24.848529Z","iopub.execute_input":"2021-06-21T15:57:24.849118Z","iopub.status.idle":"2021-06-21T15:57:24.854147Z","shell.execute_reply.started":"2021-06-21T15:57:24.849083Z","shell.execute_reply":"2021-06-21T15:57:24.853368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(train_out_file, 'w') as f:\n    json.dump(data_train, f, default=myconverter)#indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:57:28.370630Z","iopub.execute_input":"2021-06-21T15:57:28.371240Z","iopub.status.idle":"2021-06-21T15:57:28.567965Z","shell.execute_reply.started":"2021-06-21T15:57:28.371202Z","shell.execute_reply":"2021-06-21T15:57:28.566914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in range(len(vl)):\n    image_id = vl.iloc[idx].id\n    class_id = vl.iloc[idx].label_encoded #binary_label\n    path = vl.iloc[idx].path\n    img = cv2.imread((path), cv2.IMREAD_COLOR)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)#.astype(np.float32)\n    width = im_size #int(vl.iloc[idx].w)\n    height = im_size #int(vl.iloc[idx].h)\n    data_val['images'].append(dict(\n            file_name = image_id+'.png',\n            width = width,\n            height = height,\n            date_captured=None,\n            id=int(idx)))\n\n    try:\n        a = vl.iloc[idx].boxes\n        boxes = ast.literal_eval(a)\n        for box in boxes:\n            x1= float(box['x'])\n            y1 = float(box['y'])\n            w = float(box['width'])\n            h = float(box['height'])\n            bb = []\n            bbox = [\n                    int(x1),\n                    int(y1),\n                    int(w),\n                    int(h)]\n            bb.append(bbox)\n            a = int(class_id)\n            b = str(class_id)\n            ct = {f'{a}:{b}'}\n            sample = transform(image=img, bboxes=bb, labels=b)\n            vl_boxes = sample['bboxes']\n            x_1 = vl_boxes[0][0]\n            y_1 = vl_boxes[0][1]\n            w_1 = vl_boxes[0][2]\n            h_1 = vl_boxes[0][3]\n            b_b = [int(x_1), int(y_1), int(w_1), int(h_1)]\n            vl_labels = sample['labels']\n            area = (w_1)*(h_1)\n            data_val['annotations'].append(dict(id=len(data_val['annotations']),\n                                                          area=round(area,3), \n                                                          bbox=b_b,\n                                                          iscrowd= 0, #1,\n                                                          image_id=int(idx),\n                                                          category_id=int(vl_labels[0])))\n                                                          #segmentation = rl ))\n    except ValueError:\n        print(image_id)\n    \n#     with open(train_out_file, 'w') as f:\n#         json.dump(data_train, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:57:52.717813Z","iopub.execute_input":"2021-06-21T15:57:52.718173Z","iopub.status.idle":"2021-06-21T15:59:21.736766Z","shell.execute_reply.started":"2021-06-21T15:57:52.718143Z","shell.execute_reply":"2021-06-21T15:59:21.735782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def myconverter(obj):\n    if isinstance(obj, np.integer):\n        return int(obj)\n    elif isinstance(obj, np.floating):\n        return float(obj)\n    elif isinstance(obj, np.ndarray):\n        return obj.tolist()\n    elif isinstance(obj, datetime.datetime):\n        return obj.__str__()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:59:56.950920Z","iopub.execute_input":"2021-06-21T15:59:56.951301Z","iopub.status.idle":"2021-06-21T15:59:56.957553Z","shell.execute_reply.started":"2021-06-21T15:59:56.951267Z","shell.execute_reply":"2021-06-21T15:59:56.956220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(val_out_file, 'w') as f:\n    json.dump(data_val, f, default=myconverter)#indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T15:59:59.109545Z","iopub.execute_input":"2021-06-21T15:59:59.109919Z","iopub.status.idle":"2021-06-21T15:59:59.164301Z","shell.execute_reply.started":"2021-06-21T15:59:59.109885Z","shell.execute_reply":"2021-06-21T15:59:59.162920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"   # JSON для оригинальных изображений\n   # JSON for original images","metadata":{}},{"cell_type":"code","source":"len(tr)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:23.320162Z","iopub.execute_input":"2021-06-16T09:45:23.320811Z","iopub.status.idle":"2021-06-16T09:45:23.32751Z","shell.execute_reply.started":"2021-06-16T09:45:23.320752Z","shell.execute_reply":"2021-06-16T09:45:23.326705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in range(len(tr)):\n    image_id = tr.iloc[idx].id\n    class_id = tr.iloc[idx].label_encoded\n    width = int(tr.iloc[idx].w)\n    height = int(tr.iloc[idx].h)\n    data_original_train['images'].append(dict(\n            file_name = image_id+'.jpg',\n            width = width,\n            height = height,\n            date_captured=None,\n            id=int(idx)))\n\n    try:\n        a = tr.iloc[idx].boxes\n        boxes = ast.literal_eval(a)\n        for box in boxes:\n            x1= float(box['x'])\n            y1 = float(box['y'])\n            w = float(box['width'])\n            h = float(box['height'])\n            bbox =[\n                    x1,\n                    y1,\n                    w,\n                    h]\n            area = (w)*(h)\n            data_original_train['annotations'].append(dict(id=len(data_original_train['annotations']),\n                                                      area=area, \n                                                      bbox=bbox,\n                                                      iscrowd= 0, #1,\n                                                      image_id=int(idx),\n                                                      category_id=int(class_id)))\n                                                      #segmentation = rl ))\n    except ValueError:\n        print(image_id)\n    \n#     with open(train_out_file, 'w') as f:\n#         json.dump(data_train, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:28.456188Z","iopub.execute_input":"2021-06-16T09:45:28.456691Z","iopub.status.idle":"2021-06-16T09:45:31.618287Z","shell.execute_reply.started":"2021-06-16T09:45:28.45666Z","shell.execute_reply":"2021-06-16T09:45:31.616677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_original_train","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def myconverter(obj):\n    if isinstance(obj, np.integer):\n        return int(obj)\n    elif isinstance(obj, np.floating):\n        return float(obj)\n    elif isinstance(obj, np.ndarray):\n        return obj.tolist()\n    elif isinstance(obj, datetime.datetime):\n        return obj.__str__()","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:43.598078Z","iopub.execute_input":"2021-06-16T09:45:43.598608Z","iopub.status.idle":"2021-06-16T09:45:43.604277Z","shell.execute_reply.started":"2021-06-16T09:45:43.598575Z","shell.execute_reply":"2021-06-16T09:45:43.603066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(train_original_out_file, 'w') as f:\n    json.dump(data_original_train, f, default=myconverter)#indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:47.078061Z","iopub.execute_input":"2021-06-16T09:45:47.078477Z","iopub.status.idle":"2021-06-16T09:45:47.313418Z","shell.execute_reply.started":"2021-06-16T09:45:47.07844Z","shell.execute_reply":"2021-06-16T09:45:47.31257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vl)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:58.016874Z","iopub.execute_input":"2021-06-16T09:45:58.017296Z","iopub.status.idle":"2021-06-16T09:45:58.023023Z","shell.execute_reply.started":"2021-06-16T09:45:58.01726Z","shell.execute_reply":"2021-06-16T09:45:58.022171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in range(len(vl)):\n    image_id = vl.iloc[idx].id\n    class_id = vl.iloc[idx].label_encoded\n    width = int(vl.iloc[idx].w)\n    height = int(vl.iloc[idx].h)\n    data_original_val['images'].append(dict(\n            file_name = image_id+'.jpg',\n            width = width,\n            height = height,\n            date_captured=None,\n            id=int(idx)))\n\n    try:\n        a = vl.iloc[idx].boxes\n        boxes = ast.literal_eval(a)\n        for box in boxes:\n            x1= float(box['x'])\n            y1 = float(box['y'])\n            w = float(box['width'])\n            h = float(box['height'])\n            bbox =[\n                    x1,\n                    y1,\n                    w,\n                    h]\n            area = (w)*(h)\n            data_original_val['annotations'].append(dict(id=len(data_original_val['annotations']),\n                                                      area=area, \n                                                      bbox=bbox,\n                                                      iscrowd= 0, #1,\n                                                      image_id=int(idx),\n                                                      category_id=int(class_id)))\n                                                      #segmentation = rl ))\n    except ValueError:\n        print(image_id)\n    \n#     with open(train_out_file, 'w') as f:\n#         json.dump(data_train, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:45:59.921541Z","iopub.execute_input":"2021-06-16T09:45:59.92223Z","iopub.status.idle":"2021-06-16T09:46:00.709284Z","shell.execute_reply.started":"2021-06-16T09:45:59.922189Z","shell.execute_reply":"2021-06-16T09:46:00.708388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_original_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def myconverter(obj):\n    if isinstance(obj, np.integer):\n        return int(obj)\n    elif isinstance(obj, np.floating):\n        return float(obj)\n    elif isinstance(obj, np.ndarray):\n        return obj.tolist()\n    elif isinstance(obj, datetime.datetime):\n        return obj.__str__()","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:46:04.19135Z","iopub.execute_input":"2021-06-16T09:46:04.192094Z","iopub.status.idle":"2021-06-16T09:46:04.198452Z","shell.execute_reply.started":"2021-06-16T09:46:04.192031Z","shell.execute_reply":"2021-06-16T09:46:04.197611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(val_original_out_file, 'w') as f:\n    json.dump(data_original_val, f, default=myconverter)#indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:46:06.920338Z","iopub.execute_input":"2021-06-16T09:46:06.920698Z","iopub.status.idle":"2021-06-16T09:46:06.98322Z","shell.execute_reply.started":"2021-06-16T09:46:06.920666Z","shell.execute_reply":"2021-06-16T09:46:06.981969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Визуализация\n# Visualization\nfrom pycocotools.coco import COCO\nimport cv2\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-05-30T15:25:47.066538Z","iopub.execute_input":"2021-05-30T15:25:47.067002Z","iopub.status.idle":"2021-05-30T15:25:47.090208Z","shell.execute_reply.started":"2021-05-30T15:25:47.066958Z","shell.execute_reply":"2021-05-30T15:25:47.089019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BOX_COLOR = (255, 0, 0) # Red\nTEXT_COLOR = (255, 255, 255) # White\n\n\ndef visualize_bbox(img, bbox, class_name, color=BOX_COLOR, thickness=2):\n    \"\"\"Visualizes a single bounding box on the image\"\"\"\n    x_min, y_min, w, h = bbox\n    x_min, x_max, y_min, y_max = int(x_min), int(x_min + w), int(y_min), int(y_min + h)\n\n    cv2.rectangle(img, (x_min, y_min), (x_max, y_max), color=color, thickness=thickness)\n\n    ((text_width, text_height), _) = cv2.getTextSize(class_name, cv2.FONT_HERSHEY_SIMPLEX, 0.35, 1)    \n    cv2.rectangle(img, (x_min, y_min - int(1.3 * text_height)), (x_min + text_width, y_min), BOX_COLOR, -1)\n    cv2.putText(\n        img,\n        text=class_name,\n        org=(x_min, y_min - int(0.3 * text_height)),\n        fontFace=cv2.FONT_HERSHEY_SIMPLEX,\n        fontScale=0.35, \n        color=TEXT_COLOR, \n        lineType=cv2.LINE_AA,\n    )\n    return img","metadata":{"execution":{"iopub.status.busy":"2021-05-30T15:25:49.042397Z","iopub.execute_input":"2021-05-30T15:25:49.0429Z","iopub.status.idle":"2021-05-30T15:25:49.052164Z","shell.execute_reply.started":"2021-05-30T15:25:49.042866Z","shell.execute_reply":"2021-05-30T15:25:49.051415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize(image, bboxes, category_ids, category_id_to_name):\n    img = image.copy()\n    for bbox, category_id in zip(bboxes, category_ids):\n        class_name = category_id_to_name[category_id]\n        img = visualize_bbox(img, bbox, class_name)\n    plt.figure(figsize=(12, 12))\n    plt.axis('off')\n    plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-05-30T15:25:51.58553Z","iopub.execute_input":"2021-05-30T15:25:51.58611Z","iopub.status.idle":"2021-05-30T15:25:51.593165Z","shell.execute_reply.started":"2021-05-30T15:25:51.586074Z","shell.execute_reply":"2021-05-30T15:25:51.592032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = './train_512_annotations.json'","metadata":{"execution":{"iopub.status.busy":"2021-05-30T15:26:01.267591Z","iopub.execute_input":"2021-05-30T15:26:01.268156Z","iopub.status.idle":"2021-05-30T15:26:01.271567Z","shell.execute_reply.started":"2021-05-30T15:26:01.268121Z","shell.execute_reply":"2021-05-30T15:26:01.270833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"font = cv2.FONT_HERSHEY_PLAIN\ncoco=COCO(json_file)\ncats = coco.loadCats(coco.getCatIds())\nids = list(sorted(coco.imgs.keys()))\nimg_id = ids[617]\nann_ids = coco.getAnnIds(imgIds=img_id) # номера аннотаций изображения с боксами и пр\ncoco_annotation = coco.loadAnns(ann_ids)\npath = coco.loadImgs(img_id)[0]['file_name']\nclasses = [obj[\"category_id\"] for obj in coco_annotation]\n\nimg = cv2.imread(f'../input/coco-512/siim_512/train_512/{path}', cv2.IMREAD_COLOR)\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nnum_objs = len(coco_annotation)\nboxes = [] # список боксов в формате вок х1,у1,х2,у2\ncats = coco.loadCats(classes)# Найти категорию по идентификатору категории\nc = []\nd = []\nfor cat in cats:\n    a = cat['id']\n    b = cat['name']\n    c.append(a)\n    d.append(b)\nct = dict(zip(c, d))\nfor i in range(num_objs):\n    xmin = coco_annotation[i]['bbox'][0]-1\n    ymin = coco_annotation[i]['bbox'][1]-1\n    xmax = coco_annotation[i]['bbox'][2]-1\n    ymax = coco_annotation[i]['bbox'][3]-1\n    box = [xmin, ymin, xmax, ymax]\n    boxes.append(box)\nvisualize(img, boxes, classes, ct)","metadata":{"execution":{"iopub.status.busy":"2021-05-30T15:27:10.096412Z","iopub.execute_input":"2021-05-30T15:27:10.09705Z","iopub.status.idle":"2021-05-30T15:27:10.432922Z","shell.execute_reply.started":"2021-05-30T15:27:10.096994Z","shell.execute_reply":"2021-05-30T15:27:10.431598Z"},"trusted":true},"execution_count":null,"outputs":[]}]}