{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom skimage.io import imread\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.util import montage\nfrom skimage.morphology import label\n\nimport gc\ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-16T14:43:39.262572Z","iopub.execute_input":"2023-04-16T14:43:39.263383Z","iopub.status.idle":"2023-04-16T14:43:41.774483Z","shell.execute_reply.started":"2023-04-16T14:43:39.263327Z","shell.execute_reply":"2023-04-16T14:43:41.773026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train and Test directories\ntrain_image_dir = '../input/airbus-ship-detection/train_v2'\ntest_image_dir = '../input/airbus-ship-detection/test_v2'","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:41.778018Z","iopub.execute_input":"2023-04-16T14:43:41.778833Z","iopub.status.idle":"2023-04-16T14:43:41.785214Z","shell.execute_reply.started":"2023-04-16T14:43:41.778778Z","shell.execute_reply":"2023-04-16T14:43:41.784054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting into train directory\ntrain_images = os.listdir(train_image_dir)\ntrain_images.sort()\ntrain_images\nprint(f\"Total of {len(train_images)} images in train directory.\\nHere is how first five train_images looks like:- {train_images[:5]}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:41.787228Z","iopub.execute_input":"2023-04-16T14:43:41.788031Z","iopub.status.idle":"2023-04-16T14:43:45.232350Z","shell.execute_reply.started":"2023-04-16T14:43:41.787971Z","shell.execute_reply":"2023-04-16T14:43:45.231276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using for loop to generate different images to understand how data looks like\nplt.figure(figsize=(15,15))\nplt.suptitle('TRAIN IMAGES\\n', weight = 'bold', fontsize = 15, color = 'r')\nfor i in range(16):\n    plt.subplot(4, 4, i+1)\n    plt.imshow(imread(train_image_dir + \"/\" + train_images[i]))\n    plt.title(f\"{train_images[i]}\", weight = 'bold')\n    plt.axis('off')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:45.234766Z","iopub.execute_input":"2023-04-16T14:43:45.235822Z","iopub.status.idle":"2023-04-16T14:43:50.031595Z","shell.execute_reply.started":"2023-04-16T14:43:45.235781Z","shell.execute_reply":"2023-04-16T14:43:50.030006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train ships segmented marks\nmasks = pd.read_csv(\"../input/airbus-ship-detection/train_ship_segmentations_v2.csv\")\nmasks.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:50.033334Z","iopub.execute_input":"2023-04-16T14:43:50.034401Z","iopub.status.idle":"2023-04-16T14:43:51.521250Z","shell.execute_reply.started":"2023-04-16T14:43:50.034357Z","shell.execute_reply":"2023-04-16T14:43:51.519820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row_rle = ['10 1',\n          '4 1 2 0 4 1',\n          '3 1 4 0 3 1',\n          '2 1 6 0 2 1',\n          '1 1 2 0 1 1 2 0 1 1 2 0 1 1',\n          '1 1 8 0 1 1',\n          '3 1 1 0 2 1 1 0 3 1',\n          '2 1 1 0 1 1 2 0 1 1 1 0 2 1',\n          '1 1 1 0 1 1 1 0 2 1 1 0 1 1 1 0 1 1',\n          '10 1',\n          'Total']\n\npixels = [len(row.split(\" \")) for row in row_rle if row != 'Total']\nsum_pixels = np.array(pixels).sum()\npixels.append(sum_pixels)\n\ndata = {\n    'Row - RLE' : row_rle,\n    'Pixels' : pixels\n}\n\nrle_df = pd.DataFrame(data)\nrle_df.index+=1\nrle_df","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:51.522647Z","iopub.execute_input":"2023-04-16T14:43:51.522999Z","iopub.status.idle":"2023-04-16T14:43:51.540678Z","shell.execute_reply.started":"2023-04-16T14:43:51.522964Z","shell.execute_reply":"2023-04-16T14:43:51.539154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us now see how it works for Image id:- 0005d01c8. jpg we have in the mask data frame\n\n# Original image from training set\nimg_arr = imread(train_image_dir + '/' + '0005d01c8.jpg')\nplt.figure(figsize=(15, 8))\nplt.imshow(img_arr)\nplt. show ()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:51.542943Z","iopub.execute_input":"2023-04-16T14:43:51.543455Z","iopub.status.idle":"2023-04-16T14:43:51.994148Z","shell.execute_reply.started":"2023-04-16T14:43:51.543404Z","shell.execute_reply":"2023-04-16T14:43:51.992749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_arr. shape","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:51.996067Z","iopub.execute_input":"2023-04-16T14:43:51.996764Z","iopub.status.idle":"2023-04-16T14:43:52.005241Z","shell.execute_reply.started":"2023-04-16T14:43:51.996711Z","shell.execute_reply":"2023-04-16T14:43:52.003588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter out all 0005d01c8. jpg image ids and respective encoded data\n# 2 ships means 2 same image ids will be there!\nrle_0 = masks.query('ImageId==\"0005d01c8.jpg\"')['EncodedPixels']\nrle_0","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.007425Z","iopub.execute_input":"2023-04-16T14:43:52.008219Z","iopub.status.idle":"2023-04-16T14:43:52.044828Z","shell.execute_reply.started":"2023-04-16T14:43:52.008156Z","shell.execute_reply":"2023-04-16T14:43:52.043859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_lst, ct =[], 1\nfor mask in rle_0:\n    print(f\"Mask{ct} -\\n{mask}\\n\\n\")\n    mask_lst.append(mask)\n    ct+=1","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.048968Z","iopub.execute_input":"2023-04-16T14:43:52.049708Z","iopub.status.idle":"2023-04-16T14:43:52.055543Z","shell.execute_reply.started":"2023-04-16T14:43:52.049667Z","shell.execute_reply":"2023-04-16T14:43:52.054313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split and Display how the first mask in the list looks like\nsplit = mask_lst[0].split()\nprint(split)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.057109Z","iopub.execute_input":"2023-04-16T14:43:52.057794Z","iopub.status.idle":"2023-04-16T14:43:52.071306Z","shell.execute_reply.started":"2023-04-16T14:43:52.057757Z","shell.execute_reply":"2023-04-16T14:43:52.069594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Grab all the starting pixels and lenghts and convert it into integers using numpy\nstarts, lengths = [np.array (x, dtype = int) for x in (split[::2], split[1::2])]\nstarts, lengths","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.073607Z","iopub.execute_input":"2023-04-16T14:43:52.073976Z","iopub.status.idle":"2023-04-16T14:43:52.086617Z","shell.execute_reply.started":"2023-04-16T14:43:52.073940Z","shell.execute_reply":"2023-04-16T14:43:52.085129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the ending pixels.\n'''Examples:\n55010 1 ---> Starts at 56010 and ends at 56010\n56777 3 ---> Starts at 56777 and ends at 56779\n57544 6 ---> Star+s a+ 57544 and ends at 57549'''\nends = starts + lengths - 1\npd.DataFrame({\n'Starts' : starts,\n'Lengths' : lengths,\n'Ends' : ends\n}).head(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.088273Z","iopub.execute_input":"2023-04-16T14:43:52.088600Z","iopub.status.idle":"2023-04-16T14:43:52.101491Z","shell.execute_reply.started":"2023-04-16T14:43:52.088569Z","shell.execute_reply":"2023-04-16T14:43:52.100162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 1s in place of these pixels and rest should be 0\nimg = np.zeros (768*768, dtype = np.uint8)\nfor start, end in zip(starts, ends) :\n    img[start:end+1] = 1","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.103078Z","iopub.execute_input":"2023-04-16T14:43:52.103505Z","iopub.status.idle":"2023-04-16T14:43:52.109881Z","shell.execute_reply.started":"2023-04-16T14:43:52.103471Z","shell.execute_reply":"2023-04-16T14:43:52.108824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how output looks\nimg[56776:56781] # Should output 0, 1, 1, 1 ,0 as we know 56777, 56778, 56779 ---> 1 and 5676, 56780 --->","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.111207Z","iopub.execute_input":"2023-04-16T14:43:52.111544Z","iopub.status.idle":"2023-04-16T14:43:52.122494Z","shell.execute_reply.started":"2023-04-16T14:43:52.111513Z","shell.execute_reply":"2023-04-16T14:43:52.121489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Copy-Paste this idea for another ship in the image\nsplit_1 = mask_lst[1].split()\nstarts, lengths = [np.array(x, dtype = int) for x in (split_1[0:][::2], split_1[1:][::2])]\nends = starts + lengths - 1\nimg1 = np.zeros(768*768, dtype = np.uint8)\nfor start, end in zip(starts, ends):\n    img1[start:end+1] = 1","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.123812Z","iopub.execute_input":"2023-04-16T14:43:52.124376Z","iopub.status.idle":"2023-04-16T14:43:52.133562Z","shell.execute_reply.started":"2023-04-16T14:43:52.124341Z","shell.execute_reply":"2023-04-16T14:43:52.132124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reshaping both the ship masks and combining it to form the final mask!\nimg = img.reshape (768, 768)\nimg1 = img1.reshape (768, 768)\nfinal = img+img1\nprint (final, '\\n\\n', final.shape, \"\\n\\n\", final.ndim)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.135480Z","iopub.execute_input":"2023-04-16T14:43:52.136013Z","iopub.status.idle":"2023-04-16T14:43:52.148579Z","shell.execute_reply.started":"2023-04-16T14:43:52.135961Z","shell.execute_reply":"2023-04-16T14:43:52.147131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Expand dimension of this array to have only 1 channel in the mask and visualise original and final mask\nfinal = np.expand_dims (final, -1) # -1 means the last available dimenstion, in this case it is 2. Hence, on axis = 2 we will get 1.\noriginal = imread(train_image_dir+'/'+train_images[15])\nplt.figure(figsize=(15, 8))\nplt.subplot (1 ,2, 1)\nplt.title(f\"Original - Train Image, {original.shape}\")\nplt.imshow(original)\nplt.subplot (1, 2, 2)\nplt.title(f\"Mask generated from the RLE data for each ship, {final.shape}\")\nplt.imshow(final, cmap = \"Blues_r\")\nplt.tight_layout ()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.150726Z","iopub.execute_input":"2023-04-16T14:43:52.151694Z","iopub.status.idle":"2023-04-16T14:43:52.984435Z","shell.execute_reply.started":"2023-04-16T14:43:52.151641Z","shell.execute_reply":"2023-04-16T14:43:52.982845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define functions to do these tasks for all the training images\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    Input arguments\n    mask re: Mask of one ship in the train image \n    shape: Output shape of the image array\n    '''\n    s = mask_rle.split ()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    ends = starts + lengths - 1\n    img = np.zeros(shape[0]*shape[1], dtype=np. uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi+1] = 1\n    '''\n    Returns\n    Transposed array of the mask: Contains 1s and Os. 1 for ship and 0 for background\n    '''\n    return img.reshape(shape).T\ndef masks_as_image(in_mask_list):\n    '''\n    Input\n    in_mask_list: List of the masks of each ship in one whole training image\n    '''\n    all_masks = np.zeros( (768, 768), dtype = np.int16)\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    '''\n    Returns\n    Full mask of the training image whose RLE data has been passed as an input\n    '''\n    return np.expand_dims (all_masks, -1)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:52.986329Z","iopub.execute_input":"2023-04-16T14:43:52.986943Z","iopub.status.idle":"2023-04-16T14:43:52.999370Z","shell.execute_reply.started":"2023-04-16T14:43:52.986857Z","shell.execute_reply":"2023-04-16T14:43:52.997913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for num in [3, 4, 5, 6]:\n    rle_0 = masks.query (f'ImageId==\"{train_images[num-1]}\"')['EncodedPixels']\n    img_0 = masks_as_image (rle_0)\n    original = imread(train_image_dir+\"/\"+train_images[num-1])\n    plt. figure(figsize= (15, 8))\n    plt.subplot(1, 2, 1)\n    plt.title(f\"Original - Train Image {original.shape}\")\n    plt.imshow(original)\n    plt.subplot(1, 2, 2)\n    plt.title(f\"Mask generated from the RLE data for each ship {final.shape}\")\n    plt.imshow (img_0, cmap = \"Blues_r\")\n    plt.tight_layout ()\n    plt. show()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:53.000993Z","iopub.execute_input":"2023-04-16T14:43:53.001475Z","iopub.status.idle":"2023-04-16T14:43:56.583023Z","shell.execute_reply.started":"2023-04-16T14:43:53.001425Z","shell.execute_reply":"2023-04-16T14:43:56.581957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' Note that NaN values in the EncodedPixels are of floft type and everything else is a string type'''\n# Add a new feature to the masks data frame named as ship. If Encoded pixel in any row is a string, there is a ship else there isn'\nmasks['ships'] = masks['EncodedPixels'].map (lambda c_row: 1 if isinstance(c_row, str) else 0)\nmasks.head(9)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:56.584350Z","iopub.execute_input":"2023-04-16T14:43:56.585606Z","iopub.status.idle":"2023-04-16T14:43:56.712700Z","shell.execute_reply.started":"2023-04-16T14:43:56.585563Z","shell.execute_reply":"2023-04-16T14:43:56.711427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making a new data frame with unique image ids where we are summing up the ship counts\nunique_img_ids = masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index ()\nunique_img_ids.index+=1 # Incrimenting all the index by 1\nunique_img_ids.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:56.714095Z","iopub.execute_input":"2023-04-16T14:43:56.714459Z","iopub.status.idle":"2023-04-16T14:43:56.889618Z","shell.execute_reply.started":"2023-04-16T14:43:56.714424Z","shell.execute_reply":"2023-04-16T14:43:56.888288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding a new feature to unique_img_ids data frame. If ship exists in image, val is 1 else 0.\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:56.891117Z","iopub.execute_input":"2023-04-16T14:43:56.891483Z","iopub.status.idle":"2023-04-16T14:43:56.957371Z","shell.execute_reply.started":"2023-04-16T14:43:56.891449Z","shell.execute_reply":"2023-04-16T14:43:56.955710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the size of the files. Will take some time to run as there are loads of files!!!\nunique_img_ids['file_size_kb'] = unique_img_ids['ImageId'].map(lambda c_img_id: os.stat(os.path.join(train_image_dir, c_img_id)).st_size/1024)\n'''os.stat is used to get status of the specified path. Here, st_size represents size of the file in bytes. Converting it into KB!'''","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:43:56.958842Z","iopub.execute_input":"2023-04-16T14:43:56.959254Z","iopub.status.idle":"2023-04-16T14:54:08.111450Z","shell.execute_reply.started":"2023-04-16T14:43:56.959215Z","shell.execute_reply":"2023-04-16T14:54:08.110136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can get rid of any images whose size is less than 35 Kb. As some of the files are corrupted!\nunique_img_ids[unique_img_ids.file_size_kb<35].head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:08.113622Z","iopub.execute_input":"2023-04-16T14:54:08.114129Z","iopub.status.idle":"2023-04-16T14:54:08.135205Z","shell.execute_reply.started":"2023-04-16T14:54:08.114076Z","shell.execute_reply":"2023-04-16T14:54:08.133684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rle_0 = masks.query(f'ImageId==\"0318fc519.jpg\"')['EncodedPixels']\nimg_0 = masks_as_image(rle_0)\noriginal = imread(train_image_dir+\"/\"+'0318fc519.jpg')\nplt.figure(figsize=(15, 8)) \nplt.subplot (1, 2, 1)\nplt.title(f\"Original - Train Image {original.shape}\") \nplt.imshow(original) \nplt.subplot (1, 2, 2)\nplt.title(f\"Mask generated from the RLE data for each ship {final.shape}\")\nplt.imshow(img_0, cmap = \"Blues_r\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:08.136774Z","iopub.execute_input":"2023-04-16T14:54:08.137262Z","iopub.status.idle":"2023-04-16T14:54:08.861141Z","shell.execute_reply.started":"2023-04-16T14:54:08.137206Z","shell.execute_reply":"2023-04-16T14:54:08.859607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Keep the files whose size > 35 kB\nunique_img_ids = unique_img_ids[unique_img_ids.file_size_kb > 35]\nunique_img_ids.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:08.862682Z","iopub.execute_input":"2023-04-16T14:54:08.863184Z","iopub.status.idle":"2023-04-16T14:54:08.884001Z","shell.execute_reply.started":"2023-04-16T14:54:08.863132Z","shell.execute_reply":"2023-04-16T14:54:08.882879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Also, retrive the old masks data frame\nmasks.drop(['ships'], axis=1, inplace=True)\nmasks.index+=1\nmasks.head ()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:08.890230Z","iopub.execute_input":"2023-04-16T14:54:08.890871Z","iopub.status.idle":"2023-04-16T14:54:08.911655Z","shell.execute_reply.started":"2023-04-16T14:54:08.890829Z","shell.execute_reply":"2023-04-16T14:54:08.910231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train - Test split\nfrom sklearn.model_selection import train_test_split\ntrain_ids, valid_ids = train_test_split(unique_img_ids, test_size = 0.3, stratify = unique_img_ids['ships'])","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:08.913242Z","iopub.execute_input":"2023-04-16T14:54:08.913600Z","iopub.status.idle":"2023-04-16T14:54:09.154994Z","shell.execute_reply.started":"2023-04-16T14:54:08.913567Z","shell.execute_reply":"2023-04-16T14:54:09.153545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train data frame\ntrain_df = pd.merge(masks, train_ids)\n# Create test data frame\nvalid_df = pd.merge (masks, valid_ids)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.156568Z","iopub.execute_input":"2023-04-16T14:54:09.156948Z","iopub.status.idle":"2023-04-16T14:54:09.447791Z","shell.execute_reply.started":"2023-04-16T14:54:09.156909Z","shell.execute_reply":"2023-04-16T14:54:09.446496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are ~\")\nprint(train_df.shape[0], 'training masks.')\nprint(valid_df.shape[0], 'validation masks.')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.449408Z","iopub.execute_input":"2023-04-16T14:54:09.449901Z","iopub.status.idle":"2023-04-16T14:54:09.458301Z","shell.execute_reply.started":"2023-04-16T14:54:09.449851Z","shell.execute_reply":"2023-04-16T14:54:09.456925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise the ship counts\nplt.figure(figsize=(10, 6))\nsns.countplot(train_df.ships)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.460123Z","iopub.execute_input":"2023-04-16T14:54:09.460826Z","iopub.status.idle":"2023-04-16T14:54:09.675709Z","shell.execute_reply.started":"2023-04-16T14:54:09.460774Z","shell.execute_reply":"2023-04-16T14:54:09.674584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Clipping the max value of grounded_ship_count to be 7, minimum to be 0\ntrain_df['grouped_ship_count'] = train_df.ships.map(lambda x: (x+1)//2).clip(0,7)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.677032Z","iopub.execute_input":"2023-04-16T14:54:09.677386Z","iopub.status.idle":"2023-04-16T14:54:09.752406Z","shell.execute_reply.started":"2023-04-16T14:54:09.677354Z","shell.execute_reply":"2023-04-16T14:54:09.751105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check\ntrain_df.grouped_ship_count.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.753730Z","iopub.execute_input":"2023-04-16T14:54:09.754080Z","iopub.status.idle":"2023-04-16T14:54:09.764753Z","shell.execute_reply.started":"2023-04-16T14:54:09.754048Z","shell.execute_reply":"2023-04-16T14:54:09.763183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random 10 data\ntrain_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.766690Z","iopub.execute_input":"2023-04-16T14:54:09.767237Z","iopub.status.idle":"2023-04-16T14:54:09.793304Z","shell.execute_reply.started":"2023-04-16T14:54:09.767193Z","shell.execute_reply":"2023-04-16T14:54:09.791975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random Under-Sampling ships\ndef sample_ships(in_df, base_rep_val=1500):\n    '''\n    Input Args:\n    in_df - dataframe we want to apply this function\n    base_val - random sample of this value to be taken from the data frame\n    '''\n    if in_df['ships'].values[0]==0:\n        return in_df.sample(base_rep_val//3) # Random 1500/3 = 500 samples taken whose ship count is 0 in an imade\n    else:\n        return in_df.sample(base_rep_val)\n# Random IS samples raken whose shin count is nor a in an image","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.797094Z","iopub.execute_input":"2023-04-16T14:54:09.797476Z","iopub.status.idle":"2023-04-16T14:54:09.804921Z","shell.execute_reply.started":"2023-04-16T14:54:09.797441Z","shell.execute_reply":"2023-04-16T14:54:09.803426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating groups of ship counts and applying the sample ships functions to randomly undersample the ships\nbalanced_train_df = train_df.groupby('grouped_ship_count').apply(sample_ships)\nbalanced_train_df.grouped_ship_count.value_counts() \n# In each group we have total of 1500 ships except 0 as we have decreased it even more to 500","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.806617Z","iopub.execute_input":"2023-04-16T14:54:09.807008Z","iopub.status.idle":"2023-04-16T14:54:09.858389Z","shell.execute_reply.started":"2023-04-16T14:54:09.806970Z","shell.execute_reply":"2023-04-16T14:54:09.857221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explaining what we just did if still not clear\nfor i in range(8):\n    df_val_counts = balanced_train_df [balanced_train_df.grouped_ship_count==i].ships.value_counts()\n    print (f\"Data frame for grouped ship count = {i}:-\\n{df_val_counts}\\nSum of Values: - {df_val_counts.values.sum()}\\n\\n\" )","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.859704Z","iopub.execute_input":"2023-04-16T14:54:09.860029Z","iopub.status.idle":"2023-04-16T14:54:09.882000Z","shell.execute_reply.started":"2023-04-16T14:54:09.859998Z","shell.execute_reply":"2023-04-16T14:54:09.881131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nplt.figure(figsize = (15, 5))\nplt.suptitle( \"Train Data\", fontsize = 18, color = 'r', weight = 'bold')\nplt.subplot(1, 2, 1) \nsns.countplot(train_df.ships, palette = 'Set2')\nplt.title(\"Ship Counts - Before Balancing\", color = 'm', fontsize = 15)\nplt.ylabel( \"Count\", color = 'tab:pink', fontsize = 13)\nplt.xlabel(\"# Ships in an image\", color = 'tab:pink', fontsize = 13)\nplt.subplot(1, 2, 2)\nsns.countplot(balanced_train_df.ships, palette = 'Set2')\nplt.title(\"Ship Counts - After Balancing\", color = 'm', fontsize = 15)\nplt.xlabel(\"# Ships in an image\", color = 'tab:pink', fontsize = 13)\nplt.tight_layout()'''","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.883505Z","iopub.execute_input":"2023-04-16T14:54:09.883831Z","iopub.status.idle":"2023-04-16T14:54:09.891624Z","shell.execute_reply.started":"2023-04-16T14:54:09.883799Z","shell.execute_reply":"2023-04-16T14:54:09.890335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\nBATCH_SIZE = 4\nEDGE_CROP = 16\nNB_EPOCHS = 5\nGAUSSIAN_NOISE = 0.1\nUPSAMPÎE_MODE = 'SIMPLE'\nNET_SCALING = None\nIMG_SCALING = (1, 1)\nVALID_IMG_COUNT = 400\nMAX_TRAIN_STEPS = 200\n# Train batch size\n# While building the model\n# Training epochs\n# To be used in a layer in the model\n# SIMPLE ==> UpSampling2D, else Conv2DTranspose\n# Downsampling inside the network\n# Downsampling in preprocessing\n#valid batch size\n#Maximum number of steps_per_epoch in training","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.893626Z","iopub.execute_input":"2023-04-16T14:54:09.894007Z","iopub.status.idle":"2023-04-16T14:54:09.904343Z","shell.execute_reply.started":"2023-04-16T14:54:09.893964Z","shell.execute_reply":"2023-04-16T14:54:09.902853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image and Mask Generator\ndef make_image_gen(in_df, batch_size = BATCH_SIZE) :\n    '''\n    Inputs\n    in_df - data frame on which the function will be applied \n    batch_size - number of training examples in one iteration\n    '''\n    all_batches = list(in_df.groupby('ImageId'))\n    out_rgb = []\n    out_mask = []\n    while True:\n        np.random.shuffle(all_batches)\n        for c_img_id, c_masks in all_batches:\n            rgb_path = os.path. join (train_image_dir, c_img_id)\n            c_img = imread (rgb_path)\n            c_mask = masks_as_image (c_masks ['EncodedPixels'].values)\n            out_rgb += [c_img]\n            out_mask += [c_mask]\n            if len(out_rgb)>=batch_size:\n                yield np.stack(out_rgb)/255.0, np.stack(out_mask) \n                out_rgb, out_mask=[], []\n\n# Group ImageIds and create list of that dataframe\n# Image list\n# Mask list\n# Loop for every data\n# Shuffling the data\n# For img_id and msk_rle in all_batches\n# Get the img path\n# img array\n# Create mask of rle data for each ship in an ima\n# Append the current img in the out rab / ima list\n# Append the current mask in the out mask / mask list\n# If lenath of list is more or equal to batch size then\n# Yeild the scaled img array (b/w 0 and 1) and mask array (0 for bg and 1 for ship)\n# Empty the lists to create another batch","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.905494Z","iopub.execute_input":"2023-04-16T14:54:09.906271Z","iopub.status.idle":"2023-04-16T14:54:09.918913Z","shell.execute_reply.started":"2023-04-16T14:54:09.906216Z","shell.execute_reply":"2023-04-16T14:54:09.917654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate train data\ntrain_gen = make_image_gen (balanced_train_df)\n# Image and Mask\ntrain_x, train_y = next(train_gen)\n# Print the summary\nprint(f\"train_x ~\\nShape: {train_x.shape} \\nMin value: {train_x.min()}\\nMax value: {train_x.max ()}\")\nprint (f\"\\ntrain_y ~\\nShape: {train_y.shape} \\nMin value: {train_y.min()}\\nMax value: {train_y.max ()}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:09.920455Z","iopub.execute_input":"2023-04-16T14:54:09.920979Z","iopub.status.idle":"2023-04-16T14:54:10.815701Z","shell.execute_reply.started":"2023-04-16T14:54:09.920944Z","shell.execute_reply":"2023-04-16T14:54:10.814389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visulaising train batch\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape [3])], -1)\nbatch_rgb = montage_rgb(train_x)\nbatch_seg = montage(train_y[:, :, :, 0])\nbatch_overlap = mark_boundaries (batch_rgb, batch_seg.astype(int))\ntitles = [\"Images\", \"Seamentations\", \"Bounding Boxes on ships in Images\"]\ncolors = ['g', 'm', 'b']\ndisplay = [batch_rgb, batch_seg, batch_overlap]\nplt.figure(figsize= (25,10))\nfor i in range (3):\n    plt. subplot (1, 3, i+1)\n    plt.imshow(display [i])\n    plt.title(titles[i], fontsize = 18, color = colors[i])\n    plt.axis('off')\nplt.suptitle(\"Batch Visualizations\", fontsize = 20, color = 'r', weight = 'bold') # Add suptitle\nplt.tight_layout ()\n# Create montage of img\n# Create montafe of msk\n# Create bounding box around ships in img\n# Titles for subplot\n# colors to he used for title\n# What to display in subplot\n# Generate figure\n# For i = 0, 1, 2,\n3\n# Create subplot\n# Display\n# Title\n# Turn off the ayic\n#ravour for subnot","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:10.817663Z","iopub.execute_input":"2023-04-16T14:54:10.821501Z","iopub.status.idle":"2023-04-16T14:54:14.275269Z","shell.execute_reply.started":"2023-04-16T14:54:10.821425Z","shell.execute_reply":"2023-04-16T14:54:14.273928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare validation data\nvalid_x, valid_y = next(make_image_gen (valid_df, VALID_IMG_COUNT))\nprint(f\"valid_x ~\\nShape: {valid_x.shape} \\nMin value: {valid_x.min()} \\nMax value: (valid_x.max())\")\nprint (f\"Invalid_y ~\\nShape: {valid_y.shape} \\nMin value: {valid_y.min()}\\nMax value: {valid_y.max () }\")","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:14.277070Z","iopub.execute_input":"2023-04-16T14:54:14.277494Z","iopub.status.idle":"2023-04-16T14:54:32.051105Z","shell.execute_reply.started":"2023-04-16T14:54:14.277453Z","shell.execute_reply":"2023-04-16T14:54:32.049808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augmenting Data using ImageDataGenerator\nfrom keras.preprocessing.image import ImageDataGenerator\n\n# Preparina image data generator arauments\ndg_args = dict(rotation_range = 15,            # Degree range for random rotations\n               horizontal_flip = True,        # Randomly flips the inputs borizontally\n               vertical_flip = True,          # Randomly flips the inputs vertically\n               data_format = 'channels_last') # channels last refer to (batch, height, width, channels)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:32.053164Z","iopub.execute_input":"2023-04-16T14:54:32.053516Z","iopub.status.idle":"2023-04-16T14:54:41.142847Z","shell.execute_reply.started":"2023-04-16T14:54:32.053481Z","shell.execute_reply":"2023-04-16T14:54:41.141439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_gen = ImageDataGenerator(**dg_args)\nlabel_gen = ImageDataGenerator(**dg_args)\ndef create_aug_gen(in_gen, seed = None):\n    '''\n    Takes in -\n    in_gen - train data generator, seed value\n    '''\n    np.random.seed(seed if seed is not None else np.random.choice(range (9999))) # Randomly assign seed value if not provided \n    for in_x, in_y in in_gen:\n        # For imgs and msks in train data generator\n        seed = 12\n        # Seed value for imgs and msks must be same else augmentation won't be same\n        # Create augmented imgs\n        g_x = image_gen. flow(255*in_x,\n                              batch_size = in_x.shape[0],\n                              seed = seed,\n                              shuffle=True)\n    # Inverse scaling on imgs for augmentation\n    # batch_size = 3\n    # Seed\n    # Shuffle the data\n    # Create auamented masks\n        g_y = label_gen.flow(in_y,\n                             batch_size = in_x.shape[0],\n                             seed = seed,\n                             shuffle=True)\n        '''Yeilds - augmented scaled imgs and msks array'''\n        yield next(g_x)/255.0, next(g_y)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:41.145316Z","iopub.execute_input":"2023-04-16T14:54:41.146950Z","iopub.status.idle":"2023-04-16T14:54:41.157560Z","shell.execute_reply.started":"2023-04-16T14:54:41.146891Z","shell.execute_reply":"2023-04-16T14:54:41.156103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Augment the train dara\ncur_gen = create_aug_gen(train_gen, seed = 42)\nt_x, t_y = next(cur_gen)\nprint('x', t_x.shape, t_x.dtype, t_x.min(), t_x.max())\nprint('y', t_y.shape, t_y.dtype, t_y.min(), t_y.max())","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:41.159451Z","iopub.execute_input":"2023-04-16T14:54:41.160618Z","iopub.status.idle":"2023-04-16T14:54:42.121774Z","shell.execute_reply.started":"2023-04-16T14:54:41.160566Z","shell.execute_reply":"2023-04-16T14:54:42.120607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final display before passing data into model\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (25, 10))\nax1.imshow (montage_rgb(t_x), cmap='gray')\nax1.set_title('Images', fontsize = 18, color = 'g')\nax1.axis('off')\nax2.imshow(montage(t_y[:,:, :, 0]), cmap='Blues_r') \nax2.set_title('Masks', fontsize = 18, color = 'r')\nax2.axis('off')\nax3.imshow(mark_boundaries(montage_rgb(t_x), montage(t_y[:, :,:, 0].astype(int)))) \nax3.set_title ('Bounding Box', fontsize = 18, color = 'b' )\nax3.axis('off')\nplt.tight_layout ()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:42.123278Z","iopub.execute_input":"2023-04-16T14:54:42.123609Z","iopub.status.idle":"2023-04-16T14:54:45.137708Z","shell.execute_reply.started":"2023-04-16T14:54:42.123578Z","shell.execute_reply":"2023-04-16T14:54:45.136457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect () # Block all the garbage that has been generated","metadata":{"execution":{"iopub.status.busy":"2023-04-16T14:54:45.139338Z","iopub.execute_input":"2023-04-16T14:54:45.139673Z","iopub.status.idle":"2023-04-16T14:54:45.432727Z","shell.execute_reply.started":"2023-04-16T14:54:45.139641Z","shell.execute_reply":"2023-04-16T14:54:45.431401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}