{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn import preprocessing","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-15T11:18:34.170729Z","iopub.execute_input":"2022-09-15T11:18:34.171696Z","iopub.status.idle":"2022-09-15T11:18:34.774353Z","shell.execute_reply.started":"2022-09-15T11:18:34.171614Z","shell.execute_reply":"2022-09-15T11:18:34.773120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert Run length encoding to bounding boxes\n\n\n    1. implement a function to convert rle to bounding box\n    2. Save all the bounding boxes in the new csv file, alongside with its image names","metadata":{}},{"cell_type":"code","source":"!mkdir train_images/ train_labels/\n!mkdir val_images/ val_labels/\n!mkdir test_images/ test_labels/","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:18:34.777306Z","iopub.execute_input":"2022-09-15T11:18:34.777865Z","iopub.status.idle":"2022-09-15T11:18:38.219687Z","shell.execute_reply.started":"2022-09-15T11:18:34.777832Z","shell.execute_reply":"2022-09-15T11:18:38.217828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:18:38.221823Z","iopub.execute_input":"2022-09-15T11:18:38.222724Z","iopub.status.idle":"2022-09-15T11:18:39.341696Z","shell.execute_reply.started":"2022-09-15T11:18:38.222661Z","shell.execute_reply":"2022-09-15T11:18:39.340522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rletobbox(rle, shape, image_filename, counter):\n    '''\n    rle: run length encoded image mask as string\n    shape: (height, width) of an image in which RLE was produced\n    image_filename: name of image whose bounding boxes we are going to extract\n    \n    Returns:\n        (x0, y0, x1, y1) tuple describing the bounding box of RLE mask\n    '''\n    \n    a = np.fromiter(rle.split(), dtype=np.uint)\n    a = a.reshape((-1, 2)) # an array of start, length pairs\n    a[:, 0] -= 1  # 'start' is 1-indexed\n    \n    y0 = a[:,0] % shape[0]\n    y1 = y0 + a[:,1]\n    \n    if np.any(y1 > shape[0]):\n        # got 'y' overrun, meaning that there are pixels in a mask on 0 and shape[0] position\n        y0 = 0\n        y1 = shape[0]\n    else:\n        y0 = np.min(y0)\n        y1 = np.max(y1)\n        \n    x0 = a[:, 0] // shape[0]\n    x1 = (a[:, 0] + a[:, 1]) // shape[0]\n    x0 = np.min(x0)\n    x1 = np.max(x1)\n    \n    bounding_boxes = np.array([x0, y0, x1, y1])\n    normalized_array = preprocessing.normalize([bounding_boxes])\n    \n    x0 = normalized_array[0][0]\n    y0 = normalized_array[0][1]\n    x1 = normalized_array[0][2]\n    y1 = normalized_array[0][3]\n    \n    if x1 > shape[1]:\n        # just went out of image dimension\n        raise ValueError(\"Invalid RLE or image dimensions: x1=%d > shape[1]=%d\" % (x1, shape[1]))\n    \n    if counter <= 8000:\n        with open('./val_labels/'+image_filename+'.txt', 'w') as f:\n            f.write(f'0 {x0} {y0} {x1} {y1}')\n        #print(\"file created sucessfully\")\n    elif(counter > 8000 and counter <= 16000):\n        with open('./test_labels/'+image_filename+'.txt', 'w') as f:\n            f.write(f'0 {x0} {y0} {x1} {y1}') \n    elif(counter > 16000):\n        with open('./train_labels/'+image_filename+'.txt', 'w') as f:\n            f.write(f'0 {x0} {y0} {x1} {y1}')\n        #print(\"file created sucessfully\")","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:18:39.344333Z","iopub.execute_input":"2022-09-15T11:18:39.344706Z","iopub.status.idle":"2022-09-15T11:18:39.360234Z","shell.execute_reply.started":"2022-09-15T11:18:39.344670Z","shell.execute_reply":"2022-09-15T11:18:39.358837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image, ImageDraw\nimport shutil, os\n\ncounter = 1\n\ndf = pd.read_csv('../input/airbus-ship-detection/train_ship_segmentations_v2.csv')\nfor _, image_filename, encoded_pixel in df[~df.EncodedPixels.isnull()].itertuples():\n    image = Image.open('../input/airbus-ship-detection/train_v2/'+image_filename)\n    image_bbox_filename = os.path.splitext(image_filename)[0]\n    rletobbox(encoded_pixel, (image.height, image.width), image_bbox_filename, counter)\n    \n    if (counter <= 8000):\n        shutil.copy('../input/airbus-ship-detection/train_v2/'+image_filename, './val_images')\n        print(\"file copied scuesfully to val\", counter)\n        \n    elif(counter > 8000 and counter <= 16000):\n        shutil.copy('../input/airbus-ship-detection/train_v2/'+image_filename, './test_images')\n        print(\"file copied scuesfully to test\", counter)\n    elif(counter > 16000):\n        shutil.copy('../input/airbus-ship-detection/train_v2/'+image_filename, './train_images')\n        print(\"file copied scuesfully to train\", counter)\n        \n    counter += 1","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-09-15T11:18:39.363072Z","iopub.execute_input":"2022-09-15T11:18:39.363467Z","iopub.status.idle":"2022-09-15T11:28:24.145263Z","shell.execute_reply.started":"2022-09-15T11:18:39.363430Z","shell.execute_reply":"2022-09-15T11:28:24.143411Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"e\n!ls ./train_images | wc -l\n!ls ./train_labels | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:28:24.148021Z","iopub.execute_input":"2022-09-15T11:28:24.148440Z","iopub.status.idle":"2022-09-15T11:28:26.598815Z","shell.execute_reply.started":"2022-09-15T11:28:24.148405Z","shell.execute_reply":"2022-09-15T11:28:26.597256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ./val_images | wc -l\n!ls ./val_labels | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:28:26.601103Z","iopub.execute_input":"2022-09-15T11:28:26.601621Z","iopub.status.idle":"2022-09-15T11:28:28.894484Z","shell.execute_reply.started":"2022-09-15T11:28:26.601580Z","shell.execute_reply":"2022-09-15T11:28:28.893156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ./test_images | wc -l\n!ls ./test_labels | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:28:28.896762Z","iopub.execute_input":"2022-09-15T11:28:28.897124Z","iopub.status.idle":"2022-09-15T11:28:31.158600Z","shell.execute_reply.started":"2022-09-15T11:28:28.897086Z","shell.execute_reply":"2022-09-15T11:28:31.157103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cd /kaggle/working\n!tar -czvf images-labels.tar -C . .","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-09-15T11:28:31.160657Z","iopub.execute_input":"2022-09-15T11:28:31.161038Z","iopub.status.idle":"2022-09-15T11:35:39.303197Z","shell.execute_reply.started":"2022-09-15T11:28:31.161001Z","shell.execute_reply":"2022-09-15T11:35:39.301688Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:35:39.306945Z","iopub.execute_input":"2022-09-15T11:35:39.308085Z","iopub.status.idle":"2022-09-15T11:35:40.483205Z","shell.execute_reply.started":"2022-09-15T11:35:39.308039Z","shell.execute_reply":"2022-09-15T11:35:40.481655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r train_images\n!rm -r train_labels\n!rm -r val_images\n!rm -r val_labels\n!rm -r test_images\n!rm -r test_labels","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:35:40.485251Z","iopub.execute_input":"2022-09-15T11:35:40.485663Z","iopub.status.idle":"2022-09-15T11:35:50.523824Z","shell.execute_reply.started":"2022-09-15T11:35:40.485626Z","shell.execute_reply":"2022-09-15T11:35:50.522264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-09-15T11:35:50.526072Z","iopub.execute_input":"2022-09-15T11:35:50.526932Z","iopub.status.idle":"2022-09-15T11:35:51.672691Z","shell.execute_reply.started":"2022-09-15T11:35:50.526879Z","shell.execute_reply":"2022-09-15T11:35:51.671358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}