{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":9988,"databundleVersionId":868324,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom skimage.io import imread\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.util import montage\nfrom skimage.morphology import label\n\nimport gc\ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:06.316921Z","iopub.execute_input":"2025-08-18T12:14:06.317288Z","iopub.status.idle":"2025-08-18T12:14:06.324101Z","shell.execute_reply.started":"2025-08-18T12:14:06.317265Z","shell.execute_reply":"2025-08-18T12:14:06.32294Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"exploring the data","metadata":{}},{"cell_type":"code","source":"# Train and test directories\ntrain_image_dir = '../input/airbus-ship-detection/train_v2'\ntest_image_dir = \"../input/airbus-ship-detection/test_v2\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:06.325526Z","iopub.execute_input":"2025-08-18T12:14:06.325878Z","iopub.status.idle":"2025-08-18T12:14:06.347728Z","shell.execute_reply.started":"2025-08-18T12:14:06.325854Z","shell.execute_reply":"2025-08-18T12:14:06.346676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting into train directory\ntrain_images = os.listdir(train_image_dir)\ntrain_images.sort()\nprint(f\"Total of {len(train_images)} images in train directory.\\nHere is how first five train_images looks like:- {train_images[:5]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:06.349176Z","iopub.execute_input":"2025-08-18T12:14:06.349455Z","iopub.status.idle":"2025-08-18T12:14:08.558191Z","shell.execute_reply.started":"2025-08-18T12:14:06.349424Z","shell.execute_reply":"2025-08-18T12:14:08.556933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using for loop to generate different images to understand how data looks like\nplt.figure(figsize=(15,15))\nplt.suptitle('TRAIN IMAGES\\n', weight = 'bold', fontsize = 15, color = 'r')\nfor i in range(16):\n    plt.subplot(4, 4, i+1)\n    plt.imshow(imread(train_image_dir + \"/\" + train_images[i]))\n    plt.title(f\"{train_images[i]}\", weight = 'bold')\n    plt.axis('off')\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:08.559247Z","iopub.execute_input":"2025-08-18T12:14:08.559509Z","iopub.status.idle":"2025-08-18T12:14:12.82274Z","shell.execute_reply.started":"2025-08-18T12:14:08.55948Z","shell.execute_reply":"2025-08-18T12:14:12.821359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train ships segmented masks\nmasks = pd.read_csv(\"../input/airbus-ship-detection/train_ship_segmentations_v2.csv\")\nmasks.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:12.825251Z","iopub.execute_input":"2025-08-18T12:14:12.825663Z","iopub.status.idle":"2025-08-18T12:14:14.219788Z","shell.execute_reply.started":"2025-08-18T12:14:12.825626Z","shell.execute_reply":"2025-08-18T12:14:14.218813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row_rle = ['10 1', \n           '4 1 2 0 4 1', \n           '3 1 4 0 3 1', \n           '2 1 6 0 2 1',\n           '1 1 2 0 1 1 2 0 1 1 2 0 1 1', \n           '1 1 8 0 1 1', \n           '3 1 1 0 2 1 1 0 3 1', \n           '2 1 1 0 1 1 2 0 1 1 1 0 2 1', \n           '1 1 1 0 1 1 1 0 2 1 1 0 1 1 1 0 1 1', \n           '10 1',\n           'Total']\n\npixels = [len(row.split(\" \")) for row in row_rle if row != 'Total']\nsum_pixels = np.array(pixels).sum()\npixels.append(sum_pixels)\n\ndata = {\n    'Row - RLE' : row_rle,\n    'Pixels' : pixels\n}\n\nrle_df = pd.DataFrame(data)\nrle_df.index+=1\nrle_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.220866Z","iopub.execute_input":"2025-08-18T12:14:14.221164Z","iopub.status.idle":"2025-08-18T12:14:14.234745Z","shell.execute_reply.started":"2025-08-18T12:14:14.221142Z","shell.execute_reply":"2025-08-18T12:14:14.233702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let us now see how it works for Image id:- 0005d01c8.jpg we have in the mask data frame\n\n# Original image from training set\nimg_arr = imread(train_image_dir + '/' + '0005d01c8.jpg')\nplt.figure(figsize=(15,8))\nplt.imshow(img_arr)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.235972Z","iopub.execute_input":"2025-08-18T12:14:14.236342Z","iopub.status.idle":"2025-08-18T12:14:14.636032Z","shell.execute_reply.started":"2025-08-18T12:14:14.236311Z","shell.execute_reply":"2025-08-18T12:14:14.634977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_arr.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.63714Z","iopub.execute_input":"2025-08-18T12:14:14.637469Z","iopub.status.idle":"2025-08-18T12:14:14.644114Z","shell.execute_reply.started":"2025-08-18T12:14:14.637442Z","shell.execute_reply":"2025-08-18T12:14:14.643136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter out all 0005d01c8.jpg image ids and respective encoded data \n# 2 ships means 2 same image ids will be there!\nrle_0 = masks.query('ImageId==\"0005d01c8.jpg\"')['EncodedPixels']\nrle_0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.6451Z","iopub.execute_input":"2025-08-18T12:14:14.64536Z","iopub.status.idle":"2025-08-18T12:14:14.688038Z","shell.execute_reply.started":"2025-08-18T12:14:14.645341Z","shell.execute_reply":"2025-08-18T12:14:14.68696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a list of each mask shown above also visualise whats happening!\nmask_lst, ct = [], 1\nfor mask in rle_0:\n    print(f\"Mask {ct} -\\n{mask}\\n\\n\")\n    mask_lst.append(mask)\n    ct+=1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.692509Z","iopub.execute_input":"2025-08-18T12:14:14.692919Z","iopub.status.idle":"2025-08-18T12:14:14.707592Z","shell.execute_reply.started":"2025-08-18T12:14:14.692891Z","shell.execute_reply":"2025-08-18T12:14:14.706268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split and Display how the first mask in the list looks like\nsplit = mask_lst[0].split()\nprint(split)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.708857Z","iopub.execute_input":"2025-08-18T12:14:14.709184Z","iopub.status.idle":"2025-08-18T12:14:14.732769Z","shell.execute_reply.started":"2025-08-18T12:14:14.70916Z","shell.execute_reply":"2025-08-18T12:14:14.731791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Grab all the starting pixels and lenghts and convert it into integers using numpy \nstarts, lengths = [np.array(x, dtype = int) for x in (split[::2], split[1::2])]\nstarts, lengths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.733932Z","iopub.execute_input":"2025-08-18T12:14:14.734257Z","iopub.status.idle":"2025-08-18T12:14:14.759174Z","shell.execute_reply.started":"2025-08-18T12:14:14.734236Z","shell.execute_reply":"2025-08-18T12:14:14.758076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the ending pixels. \n'''Examples:- \n56010 1 ---> Starts at 56010 and ends at 56010\n56777 3 ---> Starts at 56777 and ends at 56779\n57544 6 ---> Starts at 57544 and ends at 57549'''\nends = starts + lengths - 1\npd.DataFrame({\n    'Starts' : starts,\n    'Lengths' : lengths,\n    'Ends' : ends\n}).head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.760417Z","iopub.execute_input":"2025-08-18T12:14:14.760763Z","iopub.status.idle":"2025-08-18T12:14:14.80086Z","shell.execute_reply.started":"2025-08-18T12:14:14.760736Z","shell.execute_reply":"2025-08-18T12:14:14.798474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create 1s in place of these pixels and rest should be 0\nimg = np.zeros(768*768, dtype = np.uint8)\nfor start, end in zip(starts, ends):\n    img[start:end+1] = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.803204Z","iopub.execute_input":"2025-08-18T12:14:14.803563Z","iopub.status.idle":"2025-08-18T12:14:14.831538Z","shell.execute_reply.started":"2025-08-18T12:14:14.803534Z","shell.execute_reply":"2025-08-18T12:14:14.829485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check how output looks\nimg[56776:56781] # Should output 0, 1 , 1, 1 ,0 as we know 56777, 56778, 56779 ---> 1 and 5676, 56780 ---> 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.832907Z","iopub.execute_input":"2025-08-18T12:14:14.833913Z","iopub.status.idle":"2025-08-18T12:14:14.857946Z","shell.execute_reply.started":"2025-08-18T12:14:14.833877Z","shell.execute_reply":"2025-08-18T12:14:14.85682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy-Paste this idea for another ship in the image\nsplit_1 = mask_lst[1].split()                                                                # Split the mask into start_pixels and lengths\nstarts, lengths = [np.array(x, dtype = int) for x in (split_1[0:][::2], split_1[1:][::2])]   # Generate arrays from only starts and lengths\nends = starts + lengths - 1                                                                  # Start pixel to end pixel will be start - 1 + length\nimg1 = np.zeros(768*768, dtype = np.uint8)                                                   # 1D array containing all zeros\nfor start, end in zip(starts, ends):                                                         # For each start to end pair\n    img1[start:end+1] = 1                                                                    # Convert the values from 0 to 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.859089Z","iopub.execute_input":"2025-08-18T12:14:14.859423Z","iopub.status.idle":"2025-08-18T12:14:14.880296Z","shell.execute_reply.started":"2025-08-18T12:14:14.859388Z","shell.execute_reply":"2025-08-18T12:14:14.879194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reshaping both the ship masks and combining it to form the final mask!\nimg = img.reshape(768, 768)\nimg1 = img1.reshape(768, 768)\nfinal = img+img1\nprint(final, '\\n\\n', final.shape, '\\n\\n', final.ndim)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.881447Z","iopub.execute_input":"2025-08-18T12:14:14.881774Z","iopub.status.idle":"2025-08-18T12:14:14.914974Z","shell.execute_reply.started":"2025-08-18T12:14:14.881748Z","shell.execute_reply":"2025-08-18T12:14:14.913906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Expand dimension of this array to have only 1 channel in the mask and visualise original and final mask\nfinal = np.expand_dims(final, -1) # -1 means the last available dimenstion, in this case it is 2. Hence, on axis = 2 we will get 1.\noriginal = imread(train_image_dir+'/'+train_images[15])\nplt.figure(figsize=(15, 8))\nplt.subplot(1, 2, 1)\nplt.title(f\"Original - Train Image, {original.shape}\")\nplt.imshow(original)\nplt.subplot(1, 2, 2)\nplt.title(f\"Mask generated from the RLE data for each ship, {final.shape}\")\nplt.imshow(final, cmap = \"gray\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:14.915693Z","iopub.execute_input":"2025-08-18T12:14:14.916046Z","iopub.status.idle":"2025-08-18T12:14:15.733574Z","shell.execute_reply.started":"2025-08-18T12:14:14.915994Z","shell.execute_reply":"2025-08-18T12:14:15.732523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy Paste the code from the prev cell with one change - Transpose!\nimg = img.reshape(768, 768).T     # Transpose the first ship mask\nimg1 = img1.reshape(768, 768).T   # Transpose the second ship mask\nfinal = img+img1                  # Generate the final mask with two ships \nfinal = np.expand_dims(final, -1) \nplt.figure(figsize=(15, 8))\nplt.subplot(1, 2, 1)\nplt.title(f\"Original - Train Image, {original.shape}\")\nplt.imshow(original)\nplt.subplot(1, 2, 2)\nplt.title(f\"Mask generated from the RLE data for each ship, {final.shape}\")\nplt.imshow(final, cmap = \"Blues_r\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:15.7345Z","iopub.execute_input":"2025-08-18T12:14:15.734776Z","iopub.status.idle":"2025-08-18T12:14:16.515887Z","shell.execute_reply.started":"2025-08-18T12:14:15.734753Z","shell.execute_reply":"2025-08-18T12:14:16.514763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define functions to do these tasks for all the training images\ndef rle_decode(mask_rle, shape=(768,768)):\n    '''\n    Input arguments -\n    mask_rle: Mask of one ship in the train image\n    shape: Output shape of the image array\n    '''\n    s = mask_rle.split()                                                               # Split the mask of each ship that is in RLE format\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]     # Get the start pixels and lengths for which image has ship\n    ends = starts + lengths - 1                                                        # Get the end pixels where we need to stop\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)                                  # A 1D vec full of zeros of size = 768*768\n    for lo, hi in zip(starts, ends):                                                   # For each start to end pixels where ship exists\n        img[lo:hi+1] = 1                                                               # Fill those values with 1 in the main 1D vector\n    '''\n    Returns -\n    Transposed array of the mask: Contains 1s and 0s. 1 for ship and 0 for background\n    '''\n    return img.reshape(shape).T                                                       \n\ndef masks_as_image(in_mask_list):\n    '''\n    Input - \n    in_mask_list: List of the masks of each ship in one whole training image\n    '''\n    all_masks = np.zeros((768, 768), dtype = np.int16)                                 # Creating 0s for the background\n    for mask in in_mask_list:                                                          # For each ship rle data in the list of mask rle \n        if isinstance(mask, str):                                                      # If the datatype is string\n            all_masks += rle_decode(mask)                                              # Use rle_decode to create one mask for whole image\n    '''\n    Returns - \n    Full mask of the training image whose RLE data has been passed as an input\n    '''\n    return np.expand_dims(all_masks, -1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:16.516909Z","iopub.execute_input":"2025-08-18T12:14:16.517185Z","iopub.status.idle":"2025-08-18T12:14:16.526947Z","shell.execute_reply.started":"2025-08-18T12:14:16.517164Z","shell.execute_reply":"2025-08-18T12:14:16.525067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for num in [3, 4, 5, 6]:\n    rle_0 = masks.query(f'ImageId==\"{train_images[num-1]}\"')['EncodedPixels']\n    img_0 = masks_as_image(rle_0)\n    original = imread(train_image_dir+\"/\"+train_images[num-1])\n    plt.figure(figsize=(15, 8))\n    plt.subplot(1, 2, 1)\n    plt.title(f\"Original - Train Image {original.shape}\")\n    plt.imshow(original)\n    plt.subplot(1, 2, 2)\n    plt.title(f\"Mask generated from the RLE data for each ship {final.shape}\")\n    plt.imshow(img_0, cmap = \"Blues_r\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:16.527961Z","iopub.execute_input":"2025-08-18T12:14:16.528302Z","iopub.status.idle":"2025-08-18T12:14:20.010254Z","shell.execute_reply.started":"2025-08-18T12:14:16.528277Z","shell.execute_reply":"2025-08-18T12:14:20.009113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''Note that NaN values in the EncodedPixels are of float type and everything else is a string type'''   \n\n# Add a new feature to the masks data frame named as ship. If Encoded pixel in any row is a string, there is a ship else there isn't. \nmasks['ships'] = masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nmasks.head(9)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:20.011406Z","iopub.execute_input":"2025-08-18T12:14:20.011665Z","iopub.status.idle":"2025-08-18T12:14:20.131305Z","shell.execute_reply.started":"2025-08-18T12:14:20.011645Z","shell.execute_reply":"2025-08-18T12:14:20.130038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Making a new data frame with unique image ids where we are summing up the ship counts\nunique_img_ids = masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index() \nunique_img_ids.index+=1 # Incrimenting all the index by 1\nunique_img_ids.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:20.132342Z","iopub.execute_input":"2025-08-18T12:14:20.13267Z","iopub.status.idle":"2025-08-18T12:14:20.308636Z","shell.execute_reply.started":"2025-08-18T12:14:20.132641Z","shell.execute_reply":"2025-08-18T12:14:20.307625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Adding two new features to unique_img_ids data frame. If ship exists in image, val is 1 else 0. And it's vec form\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:20.309694Z","iopub.execute_input":"2025-08-18T12:14:20.310052Z","iopub.status.idle":"2025-08-18T12:14:20.364349Z","shell.execute_reply.started":"2025-08-18T12:14:20.310019Z","shell.execute_reply":"2025-08-18T12:14:20.363342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the size of the files. Will take some time to run as there are loads of files!!!\nunique_img_ids['file_size_kb'] = unique_img_ids['ImageId'].map(lambda c_img_id: os.stat(os.path.join(train_image_dir, c_img_id)).st_size/1024)\n'''os.stat is used to get status of the specified path. Here, st_size represents size of the file in bytes. Converting it into kB!'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:14:20.368796Z","iopub.execute_input":"2025-08-18T12:14:20.369139Z","iopub.status.idle":"2025-08-18T12:21:55.898633Z","shell.execute_reply.started":"2025-08-18T12:14:20.369114Z","shell.execute_reply":"2025-08-18T12:21:55.897454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We can get rid of any images whose size is less than 35 Kb. As some of the files are corrupted! \nunique_img_ids[unique_img_ids.file_size_kb<35].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:55.899753Z","iopub.execute_input":"2025-08-18T12:21:55.900137Z","iopub.status.idle":"2025-08-18T12:21:55.914478Z","shell.execute_reply.started":"2025-08-18T12:21:55.900108Z","shell.execute_reply":"2025-08-18T12:21:55.913186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rle_0 = masks.query(f'ImageId==\"0318fc519.jpg\"')['EncodedPixels']\nimg_0 = masks_as_image(rle_0)\noriginal = imread(train_image_dir+\"/\"+'0318fc519.jpg')\nplt.figure(figsize=(15, 8))\nplt.subplot(1, 2, 1)\nplt.title(f\"Original - Train Image {original.shape}\")\nplt.imshow(original)\nplt.subplot(1, 2, 2)\nplt.title(f\"Mask generated from the RLE data for each ship {final.shape}\")\nplt.imshow(img_0, cmap = \"Blues_r\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:55.915392Z","iopub.execute_input":"2025-08-18T12:21:55.915669Z","iopub.status.idle":"2025-08-18T12:21:56.599394Z","shell.execute_reply.started":"2025-08-18T12:21:55.915648Z","shell.execute_reply":"2025-08-18T12:21:56.598328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Keep the files whose size > 35 kB\nunique_img_ids = unique_img_ids[unique_img_ids.file_size_kb > 35]\nunique_img_ids.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:56.600784Z","iopub.execute_input":"2025-08-18T12:21:56.60117Z","iopub.status.idle":"2025-08-18T12:21:56.622786Z","shell.execute_reply.started":"2025-08-18T12:21:56.60114Z","shell.execute_reply":"2025-08-18T12:21:56.621895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Also, retrive the old masks data frame\nmasks.drop(['ships'], axis=1, inplace=True)\nmasks.index+=1 \nmasks.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:56.623859Z","iopub.execute_input":"2025-08-18T12:21:56.624181Z","iopub.status.idle":"2025-08-18T12:21:56.647966Z","shell.execute_reply.started":"2025-08-18T12:21:56.624152Z","shell.execute_reply":"2025-08-18T12:21:56.647051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train - Test split\nfrom sklearn.model_selection import train_test_split                   \ntrain_ids, valid_ids = train_test_split(unique_img_ids, test_size = 0.3, stratify = unique_img_ids['ships'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:56.649058Z","iopub.execute_input":"2025-08-18T12:21:56.649406Z","iopub.status.idle":"2025-08-18T12:21:56.769302Z","shell.execute_reply.started":"2025-08-18T12:21:56.649375Z","shell.execute_reply":"2025-08-18T12:21:56.768301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create train data frame\ntrain_df = pd.merge(masks, train_ids)\n\n# Create test data frame\nvalid_df = pd.merge(masks, valid_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:56.770203Z","iopub.execute_input":"2025-08-18T12:21:56.770429Z","iopub.status.idle":"2025-08-18T12:21:57.075241Z","shell.execute_reply.started":"2025-08-18T12:21:56.770411Z","shell.execute_reply":"2025-08-18T12:21:57.07416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"There are ~\")\nprint(train_df.shape[0], 'training masks,')\nprint(valid_df.shape[0], 'validation masks.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.076201Z","iopub.execute_input":"2025-08-18T12:21:57.07645Z","iopub.status.idle":"2025-08-18T12:21:57.081754Z","shell.execute_reply.started":"2025-08-18T12:21:57.076431Z","shell.execute_reply":"2025-08-18T12:21:57.080653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualise the ship counts\nplt.figure(figsize=(10, 6))\nsns.countplot(train_df.ships)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.082776Z","iopub.execute_input":"2025-08-18T12:21:57.083243Z","iopub.status.idle":"2025-08-18T12:21:57.223416Z","shell.execute_reply.started":"2025-08-18T12:21:57.083212Z","shell.execute_reply":"2025-08-18T12:21:57.222344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clipping the max value of grouped_ship_count to be 7, minimum to be 0\ntrain_df['grouped_ship_count'] = train_df.ships.map(lambda x: (x+1)//2).clip(0,7)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.224366Z","iopub.execute_input":"2025-08-18T12:21:57.224685Z","iopub.status.idle":"2025-08-18T12:21:57.296764Z","shell.execute_reply.started":"2025-08-18T12:21:57.224656Z","shell.execute_reply":"2025-08-18T12:21:57.295698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check\ntrain_df.grouped_ship_count.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.297795Z","iopub.execute_input":"2025-08-18T12:21:57.298194Z","iopub.status.idle":"2025-08-18T12:21:57.308182Z","shell.execute_reply.started":"2025-08-18T12:21:57.298117Z","shell.execute_reply":"2025-08-18T12:21:57.30705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Top 10 data\ntrain_df.head(10)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.309267Z","iopub.execute_input":"2025-08-18T12:21:57.309996Z","iopub.status.idle":"2025-08-18T12:21:57.33705Z","shell.execute_reply.started":"2025-08-18T12:21:57.30997Z","shell.execute_reply":"2025-08-18T12:21:57.33581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random 10 data\ntrain_df.sample(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.338291Z","iopub.execute_input":"2025-08-18T12:21:57.338648Z","iopub.status.idle":"2025-08-18T12:21:57.368435Z","shell.execute_reply.started":"2025-08-18T12:21:57.338616Z","shell.execute_reply":"2025-08-18T12:21:57.36739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Under-Sampling ships\ndef sample_ships(in_df, base_rep_val=1500):\n    '''\n    Input Args:\n    in_df - dataframe we want to apply this function\n    base_val - random sample of this value to be taken from the data frame\n    '''\n    if in_df['ships'].values[0]==0:                                                 \n        return in_df.sample(base_rep_val//3)  # Random 1500//3 = 500 samples taken whose ship count is 0 in an image \n    else:                                 \n        return in_df.sample(base_rep_val)    # Random 1500 samples taken whose ship count is not 0 in an image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.369588Z","iopub.execute_input":"2025-08-18T12:21:57.369972Z","iopub.status.idle":"2025-08-18T12:21:57.389747Z","shell.execute_reply.started":"2025-08-18T12:21:57.369947Z","shell.execute_reply":"2025-08-18T12:21:57.388891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating groups of ship counts and applying the sample_ships functions to randomly undersample the ships\nbalanced_train_df = train_df.groupby('grouped_ship_count').apply(sample_ships)\nbalanced_train_df.grouped_ship_count.value_counts() # In each group we have total of 1500 ships except 0 as we have decreased it even more to 500","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.391083Z","iopub.execute_input":"2025-08-18T12:21:57.391376Z","iopub.status.idle":"2025-08-18T12:21:57.460731Z","shell.execute_reply.started":"2025-08-18T12:21:57.391355Z","shell.execute_reply":"2025-08-18T12:21:57.459748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Explaining what we just did if still not clear\nfor i in range(8):\n    df_val_counts = balanced_train_df[balanced_train_df.grouped_ship_count==i].ships.value_counts()\n    print(f\"Data frame for grouped ship count = {i}:-\\n{df_val_counts}\\nSum of Values:- {df_val_counts.values.sum()}\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.461604Z","iopub.execute_input":"2025-08-18T12:21:57.461916Z","iopub.status.idle":"2025-08-18T12:21:57.483133Z","shell.execute_reply.started":"2025-08-18T12:21:57.461887Z","shell.execute_reply":"2025-08-18T12:21:57.482256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(15, 5))\n\n# Adjust spacing so suptitle is visible\nplt.subplots_adjust(top=0.85)\nplt.suptitle(\"Train Data\", fontsize=18, color='r', weight='bold')\n\n# Plot 1: Before balancing\nplt.subplot(1, 2, 1)\nsns.countplot(x=train_df.ships, palette='Set2')\nplt.title(\"Ship Counts - Before Balancing\", color='m', fontsize=15)\nplt.ylabel(\"Count\", color='tab:pink', fontsize=13)\nplt.xlabel(\"# Ships in an image\", color='tab:pink', fontsize=13)\n\n# Plot 2: After balancing\nplt.subplot(1, 2, 2)\nsns.countplot(x=balanced_train_df.ships, palette='Set2')\nplt.title(\"Ship Counts - After Balancing\", color='m', fontsize=15)\nplt.xlabel(\"# Ships in an image\", color='tab:pink', fontsize=13)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.484243Z","iopub.execute_input":"2025-08-18T12:21:57.48454Z","iopub.status.idle":"2025-08-18T12:21:57.925933Z","shell.execute_reply.started":"2025-08-18T12:21:57.484517Z","shell.execute_reply":"2025-08-18T12:21:57.924977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters\nBATCH_SIZE = 4                 # Train batch size\nEDGE_CROP = 16                 # While building the model\nNB_EPOCHS = 5                  # Training epochs\nGAUSSIAN_NOISE = 0.1           # To be used in a layer in the model\nUPSAMPLE_MODE = 'SIMPLE'       # SIMPLE ==> UpSampling2D, else Conv2DTranspose\nNET_SCALING = None             # Downsampling inside the network                        \nIMG_SCALING = (1, 1)           # Downsampling in preprocessing\nVALID_IMG_COUNT = 400          # Valid batch size\nMAX_TRAIN_STEPS = 200          # Maximum number of steps_per_epoch in training","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.926888Z","iopub.execute_input":"2025-08-18T12:21:57.927181Z","iopub.status.idle":"2025-08-18T12:21:57.933118Z","shell.execute_reply.started":"2025-08-18T12:21:57.927157Z","shell.execute_reply":"2025-08-18T12:21:57.932127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Image and Mask Generator\ndef make_image_gen(in_df, batch_size = BATCH_SIZE):\n    '''\n    Inputs -\n    in_df - data frame on which the function will be applied\n    batch_size - number of training examples in one iteration\n    '''\n    all_batches = list(in_df.groupby('ImageId'))                             # Group ImageIds and create list of that dataframe\n    out_rgb = []                                                             # Image list\n    out_mask = []                                                            # Mask list\n    while True:                                                              # Loop for every data\n        np.random.shuffle(all_batches)                                       # Shuffling the data\n        for c_img_id, c_masks in all_batches:                                # For img_id and msk_rle in all_batches\n            rgb_path = os.path.join(train_image_dir, c_img_id)               # Get the img path\n            c_img = imread(rgb_path)                                         # img array\n            c_mask = masks_as_image(c_masks['EncodedPixels'].values)         # Create mask of rle data for each ship in an img\n            out_rgb += [c_img]                                               # Append the current img in the out_rgb / img list\n            out_mask += [c_mask]                                             # Append the current mask in the out_mask / mask list\n            if len(out_rgb)>=batch_size:                                     # If length of list is more or equal to batch size then\n                yield np.stack(out_rgb)/255.0, np.stack(out_mask)            # Yeild the scaled img array (b/w 0 and 1) and mask array (0 for bg and 1 for ship)\n                out_rgb, out_mask=[], []                                     # Empty the lists to create another batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.934118Z","iopub.execute_input":"2025-08-18T12:21:57.934424Z","iopub.status.idle":"2025-08-18T12:21:57.958379Z","shell.execute_reply.started":"2025-08-18T12:21:57.934398Z","shell.execute_reply":"2025-08-18T12:21:57.957316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate train data \ntrain_gen = make_image_gen(balanced_train_df)\n\n# Image and Mask\ntrain_x, train_y = next(train_gen)\n\n# Print the summary\nprint(f\"train_x ~\\nShape: {train_x.shape}\\nMin value: {train_x.min()}\\nMax value: {train_x.max()}\")\nprint(f\"\\ntrain_y ~\\nShape: {train_y.shape}\\nMin value: {train_y.min()}\\nMax value: {train_y.max()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:57.959725Z","iopub.execute_input":"2025-08-18T12:21:57.960047Z","iopub.status.idle":"2025-08-18T12:21:59.056734Z","shell.execute_reply.started":"2025-08-18T12:21:57.959994Z","shell.execute_reply":"2025-08-18T12:21:59.055684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visulaising train batch\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\nbatch_rgb = montage_rgb(train_x)                                                   # Create montage of img\nbatch_seg = montage(train_y[:, :, :, 0])                                           # Create montafe of msk\nbatch_overlap = mark_boundaries(batch_rgb, batch_seg.astype(int))                  # Create bounding box around ships in img\ntitles = [\"Images\", \"Segmentations\", \"Bounding Boxes on ships in Images\"]          # Titles for subplot\ncolors = ['g', 'm', 'b']                                                           # Colors to be used for title\ndisplay = [batch_rgb, batch_seg, batch_overlap]                                    # What to display in subplot\nplt.figure(figsize=(25,10))                                                        # Generate figure \nfor i in range(3):                                                                 # For i = 0, 1, 2, 3                           \n    plt.subplot(1, 3, i+1)                                                         # Create subplot\n    plt.imshow(display[i])                                                         # Display \n    plt.title(titles[i], fontsize = 18, color = colors[i])                         # Title \n    plt.axis('off')                                                                # Turn off the axis\nplt.suptitle(\"Batch Visualizations\", fontsize = 20, color = 'r', weight = 'bold')  # Add suptitle\nplt.tight_layout()                                                                 # Layout for subplot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:21:59.057798Z","iopub.execute_input":"2025-08-18T12:21:59.058219Z","iopub.status.idle":"2025-08-18T12:22:02.80224Z","shell.execute_reply.started":"2025-08-18T12:21:59.058195Z","shell.execute_reply":"2025-08-18T12:22:02.801212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare validation data\nvalid_x, valid_y = next(make_image_gen(valid_df, VALID_IMG_COUNT))\nprint(f\"valid_x ~\\nShape: {valid_x.shape}\\nMin value: {valid_x.min()}\\nMax value: {valid_x.max()}\")\nprint(f\"\\nvalid_y ~\\nShape: {valid_y.shape}\\nMin value: {valid_y.min()}\\nMax value: {valid_y.max()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:02.803204Z","iopub.execute_input":"2025-08-18T12:22:02.803491Z","iopub.status.idle":"2025-08-18T12:22:22.9558Z","shell.execute_reply.started":"2025-08-18T12:22:02.803469Z","shell.execute_reply":"2025-08-18T12:22:22.954709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Preparing image data generator arguments\ndg_args = dict(\n    rotation_range=15,       # Random rotation in degrees\n    horizontal_flip=True,    # Random horizontal flip\n    vertical_flip=True       # Random vertical flip\n    # channels_last is default, no need to specify\n)\n\ndatagen = ImageDataGenerator(**dg_args)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:22.956792Z","iopub.execute_input":"2025-08-18T12:22:22.957105Z","iopub.status.idle":"2025-08-18T12:22:22.962399Z","shell.execute_reply.started":"2025-08-18T12:22:22.957074Z","shell.execute_reply":"2025-08-18T12:22:22.961635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_gen = ImageDataGenerator(**dg_args)\nlabel_gen = ImageDataGenerator(**dg_args)\n\ndef create_aug_gen(in_gen, seed = None):\n    '''\n    Takes in -\n    in_gen - train data generator, seed value\n    '''\n    np.random.seed(seed if seed is not None else np.random.choice(range(9999)))  # Randomly assign seed value if not provided\n    for in_x, in_y in in_gen:                                                    # For imgs and msks in train data generator\n        seed = 12                                                                # Seed value for imgs and msks must be same else augmentation won't be same\n        \n        # Create augmented imgs\n        g_x = image_gen.flow(255*in_x,                                           # Inverse scaling on imgs for augmentation                                       \n                             batch_size = in_x.shape[0],                         # batch_size = 3\n                             seed = seed,                                        # Seed\n                             shuffle=True)                                       # Shuffle the data\n        \n        # Create augmented masks\n        g_y = label_gen.flow(in_y,\n                             batch_size = in_x.shape[0],                       \n                             seed = seed,                                         \n                             shuffle=True)                                       \n        \n        '''Yeilds - augmented scaled imgs and msks array'''\n        yield next(g_x)/255.0, next(g_y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:22.963454Z","iopub.execute_input":"2025-08-18T12:22:22.963778Z","iopub.status.idle":"2025-08-18T12:22:22.987803Z","shell.execute_reply.started":"2025-08-18T12:22:22.963745Z","shell.execute_reply":"2025-08-18T12:22:22.986601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Augment the train data\ncur_gen = create_aug_gen(train_gen, seed = 42)\nt_x, t_y = next(cur_gen)\nprint('x', t_x.shape, t_x.dtype, t_x.min(), t_x.max())\nprint('y', t_y.shape, t_y.dtype, t_y.min(), t_y.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:22.98935Z","iopub.execute_input":"2025-08-18T12:22:22.989667Z","iopub.status.idle":"2025-08-18T12:22:24.018503Z","shell.execute_reply.started":"2025-08-18T12:22:22.989636Z","shell.execute_reply":"2025-08-18T12:22:24.017531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Final display before passing data into model\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (25, 10))\nax1.imshow(montage_rgb(t_x), cmap='gray')\nax1.set_title('Images', fontsize = 18, color = 'g')\nax1.axis('off')\nax2.imshow(montage(t_y[:, :, :, 0]), cmap='Blues_r')\nax2.set_title('Masks', fontsize = 18, color = 'r')\nax2.axis('off')\nax3.imshow(mark_boundaries(montage_rgb(t_x), montage(t_y[:, :, :, 0].astype(int))))\nax3.set_title('Bounding Box', fontsize = 18, color = 'b')\nax3.axis('off')\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:24.019505Z","iopub.execute_input":"2025-08-18T12:22:24.019758Z","iopub.status.idle":"2025-08-18T12:22:26.979097Z","shell.execute_reply.started":"2025-08-18T12:22:24.019731Z","shell.execute_reply":"2025-08-18T12:22:26.978057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect() # Block all the garbage that has been generated","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:26.980082Z","iopub.execute_input":"2025-08-18T12:22:26.980385Z","iopub.status.idle":"2025-08-18T12:22:27.392701Z","shell.execute_reply.started":"2025-08-18T12:22:26.980361Z","shell.execute_reply":"2025-08-18T12:22:27.391716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import models, layers\n\n# Config variables\nUPSAMPLE_MODE = 'DECONV'     # or 'SIMPLE'\nNET_SCALING = None\nGAUSSIAN_NOISE = 0.1\nEDGE_CROP = 16\n\n# Conv2DTranspose upsampling\ndef upsample_conv(filters, kernel_size, strides, padding):\n    return layers.Conv2DTranspose(filters, kernel_size, strides=strides, padding=padding)\n\n# Upsampling without Conv2DTranspose\ndef upsample_simple(filters, kernel_size, strides, padding):\n    return layers.UpSampling2D(strides)\n\n# Upsampling method choice\nif UPSAMPLE_MODE == 'DECONV':\n    upsample = upsample_conv\nelse:\n    upsample = upsample_simple\n\n# Input layer (example: 128x128 RGB images)\ninput_img = layers.Input((128, 128, 3), name='RGB_Input')\npp_in_layer = input_img\n\n# Scaling (optional)\nif NET_SCALING is not None:\n    pp_in_layer = layers.AvgPool2D(NET_SCALING)(pp_in_layer)\n\n# Noise + normalization\npp_in_layer = layers.GaussianNoise(GAUSSIAN_NOISE)(pp_in_layer)\npp_in_layer = layers.BatchNormalization()(pp_in_layer)\n\n# ----- Encoder -----\nc1 = layers.Conv2D(8, (3, 3), activation='relu', padding='same')(pp_in_layer)\nc1 = layers.Conv2D(8, (3, 3), activation='relu', padding='same')(c1)\np1 = layers.MaxPooling2D((2, 2))(c1)\n\nc2 = layers.Conv2D(16, (3, 3), activation='relu', padding='same')(p1)\nc2 = layers.Conv2D(16, (3, 3), activation='relu', padding='same')(c2)\np2 = layers.MaxPooling2D((2, 2))(c2)\n\nc3 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(p2)\nc3 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(c3)\np3 = layers.MaxPooling2D((2, 2))(c3)\n\nc4 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(p3)\nc4 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(c4)\np4 = layers.MaxPooling2D((2, 2))(c4)\n\nc5 = layers.Conv2D(128, (3, 3), activation='relu', padding='same')(p4)\nc5 = layers.Conv2D(128, (3, 3), activation='relu', padding='same')(c5)\n\n# ----- Decoder -----\nu6 = upsample(64, (2, 2), strides=(2, 2), padding='same')(c5)\nu6 = layers.concatenate([u6, c4])\nc6 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(u6)\nc6 = layers.Conv2D(64, (3, 3), activation='relu', padding='same')(c6)\n\nu7 = upsample(32, (2, 2), strides=(2, 2), padding='same')(c6)\nu7 = layers.concatenate([u7, c3])\nc7 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(u7)\nc7 = layers.Conv2D(32, (3, 3), activation='relu', padding='same')(c7)\n\nu8 = upsample(16, (2, 2), strides=(2, 2), padding='same')(c7)\nu8 = layers.concatenate([u8, c2])\nc8 = layers.Conv2D(16, (3, 3), activation='relu', padding='same')(u8)\nc8 = layers.Conv2D(16, (3, 3), activation='relu', padding='same')(c8)\n\nu9 = upsample(8, (2, 2), strides=(2, 2), padding='same')(c8)\nu9 = layers.concatenate([u9, c1])\nc9 = layers.Conv2D(8, (3, 3), activation='relu', padding='same')(u9)\nc9 = layers.Conv2D(8, (3, 3), activation='relu', padding='same')(c9)\n\n# Output layer\nd = layers.Conv2D(1, (1, 1), activation='sigmoid')(c9)\nd = layers.Cropping2D((EDGE_CROP, EDGE_CROP))(d)\nd = layers.ZeroPadding2D((EDGE_CROP, EDGE_CROP))(d)\n\nif NET_SCALING is not None:\n    d = layers.UpSampling2D(NET_SCALING)(d)\n\n# Model\nseg_model = models.Model(inputs=[input_img], outputs=[d])\nseg_model.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:27.393907Z","iopub.execute_input":"2025-08-18T12:22:27.394237Z","iopub.status.idle":"2025-08-18T12:22:27.675705Z","shell.execute_reply.started":"2025-08-18T12:22:27.394215Z","shell.execute_reply":"2025-08-18T12:22:27.674699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute dice coefficient, loss with BCE and compile the model\nimport keras.backend as K\nfrom tensorflow.keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy\n\n# Dice coeff\ndef dice_coef(y_true, y_pred, smooth=1):\n    intersection = K.sum(y_true * y_pred, axis=[1,2,3]  )                         # int = y_true ∩ y_pred\n    union = K.sum(y_true, axis=[1,2,3]) + K.sum(y_pred, axis=[1,2,3])           # un = y_true_flatten ed ∪ y_pred_flattened\n    return K.mean( (2. * intersection + smooth) / (union + smooth), axis=0)     # dice = 2 * int + 1 / un + 1\n\n# Dice with BCE\ndef dice_p_bce(y_true, y_pred):\n    '''\n    Compute this function based on the explanation\n    - use alpha = 1e-3\n    '''\n    \n    combo_loss = \"Something\"\n         \n    return combo_loss\n\n# Compile the model\nseg_model.compile(optimizer=Adam(1e-4, decay=1e-6), loss=dice_p_bce, metrics=[dice_coef])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:27.676751Z","iopub.execute_input":"2025-08-18T12:22:27.67761Z","iopub.status.idle":"2025-08-18T12:22:27.691885Z","shell.execute_reply.started":"2025-08-18T12:22:27.677585Z","shell.execute_reply":"2025-08-18T12:22:27.691067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preparing Callbacks \nfrom keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\n\n# Best model weights\nweight_path=\"{}_weights.best.hdf5\".format('seg_model')\n\n# Preparing Callbacks \nfrom keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\n\n# Best model weights\nweight_path = \"{}.weights.best.h5\".format('seg_model')   # <-- FIXED\n\n# Monitor validation dice coeff and save the best model weights\ncheckpoint = ModelCheckpoint(weight_path, monitor='val_dice_coef', verbose=1, \n                             save_best_only=True, mode='max', save_weights_only=True)\n\n# Reduce Learning Rate on Plateau\nreduceLROnPlat = ReduceLROnPlateau(monitor='val_dice_coef', factor=0.5, \n                                   patience=3, \n                                   verbose=1, mode='max', min_delta=0.0001,  # `epsilon` → `min_delta`\n                                   cooldown=2, min_lr=1e-6)\n\n# Stop training once there is no improvement seen in the model\nearly = EarlyStopping(monitor=\"val_dice_coef\", \n                      mode=\"max\", \n                      patience=15)\n\n# Callbacks ready\ncallbacks_list = [checkpoint, early, reduceLROnPlat]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:27.692866Z","iopub.execute_input":"2025-08-18T12:22:27.693184Z","iopub.status.idle":"2025-08-18T12:22:27.727879Z","shell.execute_reply.started":"2025-08-18T12:22:27.693161Z","shell.execute_reply":"2025-08-18T12:22:27.726583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finalizing steps per epoch\nstep_count = min(MAX_TRAIN_STEPS, balanced_train_df.shape[0]//BATCH_SIZE)\n\n# Final augmented data being used in training\naug_gen = create_aug_gen(make_image_gen(balanced_train_df))\n\n# Save loss history while training\nloss_history = [seg_model.fit_generator(aug_gen, \n                             steps_per_epoch=step_count, \n                             epochs=NB_EPOCHS, \n                             validation_data=(valid_x, valid_y),\n                             callbacks=callbacks_list,\n                            workers=1)] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:27.728381Z","iopub.status.idle":"2025-08-18T12:22:27.728685Z","shell.execute_reply.started":"2025-08-18T12:22:27.728513Z","shell.execute_reply":"2025-08-18T12:22:27.728523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the weights to load it later for test data \nseg_model.load_weights(weight_path)\nseg_model.save('seg_model.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T12:22:27.730386Z","iopub.status.idle":"2025-08-18T12:22:27.730763Z","shell.execute_reply.started":"2025-08-18T12:22:27.730564Z","shell.execute_reply":"2025-08-18T12:22:27.73058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}