{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Prep Only\n\n* This notebook prepares the COTS dataset using a 5-fold group split (by 'sequence'); then selects group '4' as the validation dataset\n* Outputs YOLO format (suitable for YOLOv4, YOLOR and ScaledYOLOv4)","metadata":{}},{"cell_type":"markdown","source":"### Importing required modules","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\nimport ast\nimport os\nimport numpy as np\nimport shutil\nimport sys\nsys.path.append('../input/tensorflow-great-barrier-reef')\n\nimport cv2\nimport matplotlib.pyplot as plt\nfrom joblib import Parallel, delayed\n\nfrom sklearn.model_selection import GroupKFold\n\nimport torch\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:00:31.981069Z","iopub.execute_input":"2021-12-26T23:00:31.981612Z","iopub.status.idle":"2021-12-26T23:00:34.738981Z","shell.execute_reply.started":"2021-12-26T23:00:31.981472Z","shell.execute_reply":"2021-12-26T23:00:34.738077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Utility Functions","metadata":{}},{"cell_type":"code","source":"# hide\n\ndef coco2yolo(image_height, image_width, bboxes):\n    \"\"\"\n    coco => [xmin, ymin, w, h]\n    yolo => [xmid, ymid, w, h] (normalized)\n    \"\"\"\n    \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    for bbox in bboxes:\n        if bbox[0] + bbox[2] >= image_width:\n            bbox[2] = image_width - bbox[0] - 1\n        if bbox[1] + bbox[3] >= image_height:\n            bbox[3] = image_height - bbox[1] - 1\n    \n    # normolizinig\n    bboxes[..., [0, 2]]= bboxes[..., [0, 2]]/ image_width\n    bboxes[..., [1, 3]]= bboxes[..., [1, 3]]/ image_height\n    \n    # converstion (xmin, ymin) => (xmid, ymid)\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]/2\n    \n    return bboxes","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-26T23:00:51.676095Z","iopub.execute_input":"2021-12-26T23:00:51.676619Z","iopub.status.idle":"2021-12-26T23:00:51.686149Z","shell.execute_reply.started":"2021-12-26T23:00:51.676584Z","shell.execute_reply":"2021-12-26T23:00:51.685044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/tensorflow-great-barrier-reef'\n\ndef get_bbox(annots):\n    bboxes = [list(annot.values()) for annot in annots]\n    return bboxes\n\ndef get_path(row):\n    row['image_path'] = f'{TRAIN_PATH}/train_images/video_{row.video_id}/{row.video_frame}.jpg'\n    return row","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:01:00.302614Z","iopub.execute_input":"2021-12-26T23:01:00.303077Z","iopub.status.idle":"2021-12-26T23:01:00.309413Z","shell.execute_reply.started":"2021-12-26T23:01:00.303041Z","shell.execute_reply":"2021-12-26T23:01:00.308618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing Training Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/tensorflow-great-barrier-reef/train.csv\")\n\n# Taken only annotated photos\ndf[\"num_bbox\"] = df['annotations'].apply(lambda x: str.count(x, 'x'))\ndf_train = df[df[\"num_bbox\"]>0]\n\n#Annotations \ndf_train['annotations'] = df_train['annotations'].progress_apply(lambda x: ast.literal_eval(x))\ndf_train['bboxes'] = df_train.annotations.progress_apply(get_bbox)\n\n#Images resolution\ndf_train[\"width\"] = 1280\ndf_train[\"height\"] = 720\n\n#Path of images\ndf_train = df_train.progress_apply(get_path, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:01:01.743699Z","iopub.execute_input":"2021-12-26T23:01:01.744627Z","iopub.status.idle":"2021-12-26T23:01:06.087094Z","shell.execute_reply.started":"2021-12-26T23:01:01.74456Z","shell.execute_reply":"2021-12-26T23:01:06.086045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = GroupKFold(n_splits = 5) \ndf_train = df_train.reset_index(drop=True)\ndf_train['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(kf.split(df_train, y = df_train.video_id.tolist(), groups=df_train.sequence)):\n    df_train.loc[val_idx, 'fold'] = fold\n\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:01:07.887185Z","iopub.execute_input":"2021-12-26T23:01:07.887492Z","iopub.status.idle":"2021-12-26T23:01:07.93047Z","shell.execute_reply.started":"2021-12-26T23:01:07.887461Z","shell.execute_reply":"2021-12-26T23:01:07.929753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Output Dirs","metadata":{}},{"cell_type":"code","source":"IMAGE_DIR = \"/kaggle/working/images\"\nLABEL_DIR = \"/kaggle/working/labels\"","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:01:09.649199Z","iopub.execute_input":"2021-12-26T23:01:09.649464Z","iopub.status.idle":"2021-12-26T23:01:09.653996Z","shell.execute_reply.started":"2021-12-26T23:01:09.649435Z","shell.execute_reply":"2021-12-26T23:01:09.653237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p {IMAGE_DIR}\n!mkdir -p {LABEL_DIR}\n\n!mkdir -p {IMAGE_DIR + '/train'}\n!mkdir -p {IMAGE_DIR + '/valid'}\n!mkdir -p {LABEL_DIR + '/train'}\n!mkdir -p {LABEL_DIR + '/valid'}","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:01:11.85445Z","iopub.execute_input":"2021-12-26T23:01:11.855539Z","iopub.status.idle":"2021-12-26T23:01:16.497074Z","shell.execute_reply.started":"2021-12-26T23:01:11.855478Z","shell.execute_reply":"2021-12-26T23:01:16.495591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Moving Image Files","metadata":{}},{"cell_type":"markdown","source":"## Outputting label files","metadata":{}},{"cell_type":"code","source":"def writeLabels(bboxes, destfile, image_width = 1280, image_height = 720):\n    bboxes_coco  = np.array(bboxes).astype(np.float32).copy()\n    num_bbox     = len(bboxes_coco)\n    names        = ['cots'] * num_bbox\n    labels       = [0] * num_bbox\n    with open(destfile, 'w') as f:\n        if num_bbox<1:\n            annot = ''\n            f.write(annot)\n        else:\n            bboxes_yolo  = coco2yolo(image_height, image_width, bboxes_coco)\n            for bbox_idx in range(len(bboxes_yolo)):\n                annot = [str(labels[bbox_idx])]+ list(bboxes_yolo[bbox_idx].astype(str))+(['\\n'] if num_bbox!=(bbox_idx+1) else [''])\n                annot = ' '.join(annot)\n                annot = annot.strip(' ')\n                f.write(annot)\n    return ''\n\ndf1 = df_train[df_train.fold != 4]\nfor row_idx in tqdm(range(len(df1))):\n    row = df1.iloc[row_idx]\n    shutil.copyfile(row.image_path, f'{IMAGE_DIR}/train/{row.image_id}.jpg')\n    writeLabels(row.bboxes, f'{LABEL_DIR}/train/{row.image_id}.txt')\n\ndf2 = df_train[df_train.fold == 4]\nfor row_idx in tqdm(range(len(df2))):\n    row = df2.iloc[row_idx]\n    shutil.copyfile(row.image_path, f'{IMAGE_DIR}/valid/{row.image_id}.jpg')\n    writeLabels(row.bboxes, f'{LABEL_DIR}/valid/{row.image_id}.txt')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:03:42.004876Z","iopub.execute_input":"2021-12-26T23:03:42.005345Z","iopub.status.idle":"2021-12-26T23:04:58.796733Z","shell.execute_reply.started":"2021-12-26T23:03:42.005289Z","shell.execute_reply":"2021-12-26T23:04:58.795625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls {IMAGE_DIR + '/train/'} | wc -l","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:05:00.774267Z","iopub.execute_input":"2021-12-26T23:05:00.776043Z","iopub.status.idle":"2021-12-26T23:05:01.593715Z","shell.execute_reply.started":"2021-12-26T23:05:00.77596Z","shell.execute_reply":"2021-12-26T23:05:01.591917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls {LABEL_DIR + '/train/'} | wc -l","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:05:02.887145Z","iopub.execute_input":"2021-12-26T23:05:02.887497Z","iopub.status.idle":"2021-12-26T23:05:03.744444Z","shell.execute_reply.started":"2021-12-26T23:05:02.887466Z","shell.execute_reply":"2021-12-26T23:05:03.743238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls {IMAGE_DIR + '/valid/'} | wc -l","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:05:04.590156Z","iopub.execute_input":"2021-12-26T23:05:04.590483Z","iopub.status.idle":"2021-12-26T23:05:05.381849Z","shell.execute_reply.started":"2021-12-26T23:05:04.590453Z","shell.execute_reply":"2021-12-26T23:05:05.380538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls {LABEL_DIR + '/valid/'} | wc -l","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:05:06.310334Z","iopub.execute_input":"2021-12-26T23:05:06.311143Z","iopub.status.idle":"2021-12-26T23:05:07.102697Z","shell.execute_reply.started":"2021-12-26T23:05:06.311086Z","shell.execute_reply":"2021-12-26T23:05:07.101519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Prepping YAML file for dataset","metadata":{}},{"cell_type":"code","source":"!echo -e 'train: ../images/train\\nval: ../images/valid\\n\\nnc: 1\\nnames: ['cots']' > cots.yaml\n!cat 'cots.yaml'","metadata":{"execution":{"iopub.status.busy":"2021-12-26T23:05:08.600367Z","iopub.execute_input":"2021-12-26T23:05:08.600719Z","iopub.status.idle":"2021-12-26T23:05:10.203777Z","shell.execute_reply.started":"2021-12-26T23:05:08.600685Z","shell.execute_reply":"2021-12-26T23:05:10.202746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Zipping output files","metadata":{}},{"cell_type":"code","source":"shutil.make_archive(IMAGE_DIR, 'zip', 'images')\nshutil.make_archive(LABEL_DIR, 'zip', 'labels')","metadata":{"execution":{"iopub.status.busy":"2021-12-09T22:30:35.756508Z","iopub.execute_input":"2021-12-09T22:30:35.756965Z","iopub.status.idle":"2021-12-09T22:33:08.057405Z","shell.execute_reply.started":"2021-12-09T22:30:35.756921Z","shell.execute_reply":"2021-12-09T22:33:08.054512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cleanup","metadata":{}},{"cell_type":"code","source":"!rm -r {IMAGE_DIR}\n!rm -r {LABEL_DIR}","metadata":{"execution":{"iopub.status.busy":"2021-12-09T22:33:08.059421Z","iopub.execute_input":"2021-12-09T22:33:08.05969Z","iopub.status.idle":"2021-12-09T22:33:10.353076Z","shell.execute_reply.started":"2021-12-09T22:33:08.059656Z","shell.execute_reply":"2021-12-09T22:33:10.351737Z"},"trusted":true},"execution_count":null,"outputs":[]}]}