{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as A\nimport numpy as np\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport os\nimport torch\nimport importlib\nimport cv2\nimport pandas as pd\n\nimport ast\nimport shutil\nimport sys\n\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nfrom PIL import Image\nfrom IPython.display import display","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:09:57.935588Z","iopub.execute_input":"2022-02-04T17:09:57.936102Z","iopub.status.idle":"2022-02-04T17:10:01.665357Z","shell.execute_reply.started":"2022-02-04T17:09:57.935996Z","shell.execute_reply":"2022-02-04T17:10:01.664453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.read_csv(\"../input/tensorflow-great-barrier-reef/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:47:05.327261Z","iopub.execute_input":"2022-02-04T17:47:05.327579Z","iopub.status.idle":"2022-02-04T17:47:05.365642Z","shell.execute_reply.started":"2022-02-04T17:47:05.327535Z","shell.execute_reply":"2022-02-04T17:47:05.364783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take only annotated photos(starfish)\ndf = df[df.annotations != '[]'].reset_index()\nprint('Number of Starfish images used for training:', len(df))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:47:06.20358Z","iopub.execute_input":"2022-02-04T17:47:06.203866Z","iopub.status.idle":"2022-02-04T17:47:06.218784Z","shell.execute_reply.started":"2022-02-04T17:47:06.203838Z","shell.execute_reply":"2022-02-04T17:47:06.217671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_predictions(img, bboxes, scores, bbclasses, classes_dict, boxcolor = (0,0,255)):\n    outimg = img.copy()\n    for i in range(len(bboxes)):\n        box = bboxes[i]\n        cls_id = int(bbclasses[i])\n        score = scores[i]\n        x0 = int(box[0])\n        y0 = int(box[1])\n        x1 = x0 + int(box[2])\n        y1 = y0 + int(box[3])\n\n        cv2.rectangle(outimg, (x0, y0), (x1, y1), boxcolor, 2)\n        cv2.putText(outimg, '{}:{:.1f}%'.format(classes_dict[cls_id], score * 100), (x0, y0 - 3), cv2.FONT_HERSHEY_PLAIN, 0.8, boxcolor, thickness = 1)\n    return outimg","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:47:08.095026Z","iopub.execute_input":"2022-02-04T17:47:08.095294Z","iopub.status.idle":"2022-02-04T17:47:08.10383Z","shell.execute_reply.started":"2022-02-04T17:47:08.095265Z","shell.execute_reply":"2022-02-04T17:47:08.102932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.model_selection import StratifiedKFold\ndef get_bbox(annots):\n    bboxes=[list(annot.values()) for annot in annots]\n    return bboxes\n\ndef get_path(row):\n    row['image_path'] = f'{ROOT_DIR}/train_images/video_{row.video_id}/{row.video_frame}.jpg'\n    return row\n\nROOT_DIR = '../input/tensorflow-great-barrier-reef'\ndf[\"num_bbox\"] = df['annotations'].apply(lambda x: str.count(x, 'x'))\ndf_train = df\n#df_train\ndf_train['annotations'] = df_train['annotations'].progress_apply(lambda x : ast.literal_eval(x))\ndf_train['bboxes'] = df_train.annotations.progress_apply(get_bbox) \ndf_train = df_train.progress_apply(get_path,axis=1)\ndf_train\n\n#df_train = df_train.reset_index (drop=True)\n#df_train['fold'] = -1\n#for fold, (train_idx, val_idx) in enumerate(kf.split(df_train, y = df_train.video_id.tolist(), groups=df_train.sequence)):\n    #df_train.loc[val_idx,'fold'] = fold\n#df_test = df_train[df_train.fold == 4 ]\n","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:47:09.622981Z","iopub.execute_input":"2022-02-04T17:47:09.623892Z","iopub.status.idle":"2022-02-04T17:47:13.476076Z","shell.execute_reply.started":"2022-02-04T17:47:09.623848Z","shell.execute_reply":"2022-02-04T17:47:13.474995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_eg=df_train[df_train['num_bbox']==4]\ndf_train.iloc[750,:]","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:47:48.627669Z","iopub.execute_input":"2022-02-04T17:47:48.628602Z","iopub.status.idle":"2022-02-04T17:47:48.640831Z","shell.execute_reply.started":"2022-02-04T17:47:48.628558Z","shell.execute_reply":"2022-02-04T17:47:48.639795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths = df_train.image_path.tolist()\ngt_box = df_train.bboxes.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:16:43.708372Z","iopub.execute_input":"2022-02-04T17:16:43.708702Z","iopub.status.idle":"2022-02-04T17:16:43.714757Z","shell.execute_reply.started":"2022-02-04T17:16:43.708666Z","shell.execute_reply":"2022-02-04T17:16:43.714002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 750\nimage_id=0-4558\n\nimage_path = image_paths[i]\nImg = Image.open(image_path)\ndisplay(Img)","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:16:51.705882Z","iopub.execute_input":"2022-02-04T17:16:51.706669Z","iopub.status.idle":"2022-02-04T17:16:52.250954Z","shell.execute_reply.started":"2022-02-04T17:16:51.706627Z","shell.execute_reply":"2022-02-04T17:16:52.249949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#without crop\nimg = np.array(Img)\nout_image = draw_predictions(img, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\ndisplay(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:17:49.173462Z","iopub.execute_input":"2022-02-04T17:17:49.173745Z","iopub.status.idle":"2022-02-04T17:17:49.67099Z","shell.execute_reply.started":"2022-02-04T17:17:49.173716Z","shell.execute_reply":"2022-02-04T17:17:49.669764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_np.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:05:14.348759Z","iopub.execute_input":"2022-02-04T16:05:14.349325Z","iopub.status.idle":"2022-02-04T16:05:14.355751Z","shell.execute_reply.started":"2022-02-04T16:05:14.349289Z","shell.execute_reply":"2022-02-04T16:05:14.354724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#crop image\nimg_np = np.array(Img)[:640,:640]\nout_image = draw_predictions(img_np, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\ndisplay(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:19:07.617867Z","iopub.execute_input":"2022-02-04T17:19:07.618175Z","iopub.status.idle":"2022-02-04T17:19:07.80216Z","shell.execute_reply.started":"2022-02-04T17:19:07.618142Z","shell.execute_reply":"2022-02-04T17:19:07.801555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Albumentation","metadata":{}},{"cell_type":"code","source":"def show_augmentation(img, augmentation):\n    transform = A.Compose([augmentation])\n    img_aug = transform(image=img)['image']\n    return(img_aug)","metadata":{"execution":{"iopub.status.busy":"2022-02-04T17:20:23.236014Z","iopub.execute_input":"2022-02-04T17:20:23.23634Z","iopub.status.idle":"2022-02-04T17:20:23.241596Z","shell.execute_reply.started":"2022-02-04T17:20:23.236273Z","shell.execute_reply":"2022-02-04T17:20:23.240801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Blurring","metadata":{}},{"cell_type":"markdown","source":"### Blur","metadata":{}},{"cell_type":"code","source":"AUGMENTATION = A.Blur (p =1.0)\n\nfor q in range(1):\n    img_aug = show_augmentation(img_np, AUGMENTATION)\n    out_image = draw_predictions(img_aug,gt_box[i], [1.0]* len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T15:58:10.773759Z","iopub.execute_input":"2022-02-04T15:58:10.774106Z","iopub.status.idle":"2022-02-04T15:58:11.135551Z","shell.execute_reply.started":"2022-02-04T15:58:10.77407Z","shell.execute_reply":"2022-02-04T15:58:11.1346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GaussianBlur","metadata":{}},{"cell_type":"code","source":"AUGMENTATION = A.GaussianBlur(p = 1.0)\n\nfor q in range(1):\n    img_aug = show_augmentation(img_np, AUGMENTATION)\n    out_image = draw_predictions(img_aug, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T15:59:22.138872Z","iopub.execute_input":"2022-02-04T15:59:22.139133Z","iopub.status.idle":"2022-02-04T15:59:22.466168Z","shell.execute_reply.started":"2022-02-04T15:59:22.139107Z","shell.execute_reply":"2022-02-04T15:59:22.465373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MedianBlur","metadata":{}},{"cell_type":"code","source":"AUGMENTATION = A.MedianBlur(p=1.0)\nfor q in range(1):\n    img_aug = show_augmentation(img_np, AUGMENTATION)\n    out_image = draw_predictions(img_aug, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:00:29.734628Z","iopub.execute_input":"2022-02-04T16:00:29.734948Z","iopub.status.idle":"2022-02-04T16:00:30.049207Z","shell.execute_reply.started":"2022-02-04T16:00:29.734909Z","shell.execute_reply":"2022-02-04T16:00:30.048273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUGMENTATION = A.Downscale(scale_min=0.5, scale_max=0.5, p=1.0)\nfor q in range(1):\n    img_aug = show_augmentation(img_np, AUGMENTATION)\n    out_image = draw_predictions(img_aug, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:03:33.766305Z","iopub.execute_input":"2022-02-04T16:03:33.766906Z","iopub.status.idle":"2022-02-04T16:03:33.936483Z","shell.execute_reply.started":"2022-02-04T16:03:33.766842Z","shell.execute_reply":"2022-02-04T16:03:33.935366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ImageCompression","metadata":{}},{"cell_type":"code","source":"AUGMENTATION = A.ImageCompression(quality_lower=20, quality_upper=40, p = 1.0)\nfor q in range(1):\n    img_aug = show_augmentation(img_np, AUGMENTATION)\n    out_image = draw_predictions(img_aug, gt_box[i], [1.0] * len(gt_box[i]), [0] * len(gt_box[i]), ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:06:22.085682Z","iopub.execute_input":"2022-02-04T16:06:22.08597Z","iopub.status.idle":"2022-02-04T16:06:22.2921Z","shell.execute_reply.started":"2022-02-04T16:06:22.085941Z","shell.execute_reply":"2022-02-04T16:06:22.291215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"out_image.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:06:36.214598Z","iopub.execute_input":"2022-02-04T16:06:36.21512Z","iopub.status.idle":"2022-02-04T16:06:36.221662Z","shell.execute_reply.started":"2022-02-04T16:06:36.215084Z","shell.execute_reply":"2022-02-04T16:06:36.22078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Compound Augmentations","metadata":{}},{"cell_type":"code","source":"def show_compound_augmentation(img, bboxes, labels, augmentation_list):\n    \"\"\"\n        img: a numpy array of the image\n        bboxes: COCO-format bounding boxes\n        labels: a list of labels\n        augmentation_list: a list of functions from the Albumentations library\n        see https://albumentations.ai/docs/getting_started/transforms_and_targets/\n        \n        returns: a numpy array of the augmented image\n    \"\"\"\n    transform = A.Compose(augmentation_list,\n        bbox_params=A.BboxParams(\n        format='coco',\n        label_fields=['class_labels']\n    ))\n    transformed = transform(image=img, bboxes=bboxes, class_labels = labels)\n    img_aug = transformed['image']\n    boxes = np.array([list(b) for b in transformed['bboxes']])\n    labels = np.array(transformed['class_labels'])\n    return img_aug, boxes, labels","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:19:55.419158Z","iopub.execute_input":"2022-02-04T16:19:55.419515Z","iopub.status.idle":"2022-02-04T16:19:55.426957Z","shell.execute_reply.started":"2022-02-04T16:19:55.419482Z","shell.execute_reply":"2022-02-04T16:19:55.426318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir ./outimage","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:17:19.283785Z","iopub.execute_input":"2022-02-04T16:17:19.28444Z","iopub.status.idle":"2022-02-04T16:17:20.068941Z","shell.execute_reply.started":"2022-02-04T16:17:19.284401Z","shell.execute_reply":"2022-02-04T16:17:20.067815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUGMENTATION_LIST = [\n        A.RandomRotate90(),\n        A.Flip(),\n        A.Transpose(),\n        A.OneOf([\n            A.MotionBlur(p=.2),\n            A.MedianBlur(blur_limit=3, p=0.3),\n            A.Blur(blur_limit=3, p=0.1),\n        ], p=0.2),\n        A.ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.2, rotate_limit=45, p=0.2),\n        A.OneOf([\n            A.CLAHE(clip_limit=2),\n            A.RandomBrightnessContrast(),            \n        ], p=0.3),\n        A.HueSaturationValue(p=0.3),\n    ]\n\nfor q in range(5):\n    img_aug, bboxes, labels = show_compound_augmentation(img_np, gt_box[i], [0] * len(gt_box[i]), AUGMENTATION_LIST)\n    save_img=Image.fromarray(img_aug)\n    save_img.save(f'./augimage/{image_id}_aug{q}.jpg')\n    out_image = draw_predictions(img_aug, bboxes, [1.0] * len(bboxes), labels, ['COTS'], (0,255,0))\n    display(Image.fromarray(out_image))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T16:27:40.050606Z","iopub.execute_input":"2022-02-04T16:27:40.051567Z","iopub.status.idle":"2022-02-04T16:27:41.107089Z","shell.execute_reply.started":"2022-02-04T16:27:40.051509Z","shell.execute_reply":"2022-02-04T16:27:41.106106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(index, image_id):\n    # Read the image from image id\n    image = cv2.imread(os.path.join(BASE_DIR, 'train', f'{image_id}.jpg'), cv2.IMREAD_COLOR)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    # Get the bboxes details and apply all the augmentations\n    bboxes = train_df[train_df['image_id'] == image_id][['x_min', 'y_min', 'x_max', 'y_max']].astype(np.int32).values\n    source = train_df[train_df['image_id'] == image_id]['source'].unique()[0]\n    labels = np.ones((len(bboxes), ))  # As we have only one class (wheat heads)\n    aug_result = augmentation(image=image, bboxes=bboxes, labels=labels)\n\n    aug_image = aug_result['image']\n    aug_bboxes = aug_result['bboxes']\n    \n    Image.fromarray(image).save(os.path.join(WORK_DIR, 'train', f'{image_id}.jpg'))\n    Image.fromarray(aug_image).save(os.path.join(WORK_DIR, 'train', f'{image_id}_aug.jpg'))\n\n    image_metadata = []\n    for bbox in aug_bboxes:\n        bbox = tuple(map(int, bbox))\n        image_metadata.append({\n            'image_id': f'{image_id}_aug',\n            'x_min': bbox[0],\n            'y_min': bbox[1],\n            'x_max': bbox[2],\n            'y_max': bbox[3],\n            'width': bbox[2] - bbox[0],\n            'height': bbox[3] - bbox[1],\n            'area': (bbox[2] - bbox[0]) * (bbox[3] - bbox[1]),\n            'source': source\n        })\n    return image_metadata","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_metadata = Parallel(n_jobs=8)(delayed(create_dataset)(index, image_id) for index, image_id in tqdm(enumerate(image_ids), total=len(image_ids)))\nimage_metadata = [item for sublist in image_metadata for item in sublist]","metadata":{},"execution_count":null,"outputs":[]}]}