{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":9988,"databundleVersionId":868324,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport cv2\nimport os\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"eca8ca58-abb5-4f20-a449-6a969e1e7a63","_cell_guid":"c403160a-fc52-465e-acb9-c64a35159548","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:27.362249Z","iopub.execute_input":"2023-11-22T13:21:27.362627Z","iopub.status.idle":"2023-11-22T13:21:27.367678Z","shell.execute_reply.started":"2023-11-22T13:21:27.362597Z","shell.execute_reply":"2023-11-22T13:21:27.366414Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"        print(os.path.join(dirname, filename))","metadata":{"_uuid":"7133ff35-a8e5-40bb-af88-79e2cac240d9","_cell_guid":"5c3b7f9b-cfd0-4acc-a0b3-c6de392f5739","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-22T14:23:37.618915Z","iopub.execute_input":"2023-11-22T14:23:37.619312Z","iopub.status.idle":"2023-11-22T14:24:23.850515Z","shell.execute_reply.started":"2023-11-22T14:23:37.619279Z","shell.execute_reply":"2023-11-22T14:24:23.848209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEGMENTATION_PATH = \"/kaggle/input/airbus-ship-detection/train_ship_segmentations_v2.csv\"\nTRAIN_PATH = \"/kaggle/input/airbus-ship-detection/train_v2/\"\nTEST_PATH = \"/kaggle/input/airbus-ship-detection/test_v2/\"","metadata":{"_uuid":"1e2356b5-adb6-4b1d-ad8f-94003dd58c46","_cell_guid":"21c1a513-70f4-4fa3-a839-f25faf634325","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:27.372698Z","iopub.execute_input":"2023-11-22T13:21:27.372992Z","iopub.status.idle":"2023-11-22T13:21:27.383356Z","shell.execute_reply.started":"2023-11-22T13:21:27.372954Z","shell.execute_reply":"2023-11-22T13:21:27.382158Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segmentation_df = pd.read_csv(SEGMENTATION_PATH)","metadata":{"_uuid":"94f9b719-fa83-44fc-95cf-a64db86ff5a3","_cell_guid":"55d8a338-fea8-428e-a5ce-45cb900f6856","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:27.384818Z","iopub.execute_input":"2023-11-22T13:21:27.385136Z","iopub.status.idle":"2023-11-22T13:21:28.006091Z","shell.execute_reply.started":"2023-11-22T13:21:27.38511Z","shell.execute_reply":"2023-11-22T13:21:28.005049Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segmentation_df.info()","metadata":{"_uuid":"1c596906-8f37-4400-9e9e-867a09d23e8c","_cell_guid":"8776204a-3a57-46d1-8c34-35e12a0d0ca7","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.007394Z","iopub.execute_input":"2023-11-22T13:21:28.007714Z","iopub.status.idle":"2023-11-22T13:21:28.078508Z","shell.execute_reply.started":"2023-11-22T13:21:28.007688Z","shell.execute_reply":"2023-11-22T13:21:28.077382Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shape = (768, 768) #image shape according to info about dataset","metadata":{"_uuid":"7107db8d-7e98-43b7-a555-3b7662c3932d","_cell_guid":"2b03a96c-2185-4da4-ae8d-7b29ff9b5760","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.08067Z","iopub.execute_input":"2023-11-22T13:21:28.081026Z","iopub.status.idle":"2023-11-22T13:21:28.085325Z","shell.execute_reply.started":"2023-11-22T13:21:28.08096Z","shell.execute_reply":"2023-11-22T13:21:28.08432Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrupted_images = ['6384c3e78.jpg','13703f040.jpg', '14715c06d.jpg',  '33e0ff2d5.jpg',\n                '4d4e09f2a.jpg', '877691df8.jpg', '8b909bb20.jpg', 'a8d99130e.jpg', \n                'ad55c3143.jpg', 'c8260c541.jpg', 'd6c7f17c7.jpg', 'dc3e7c901.jpg',\n                'e44dffe88.jpg', 'ef87bad36.jpg', 'f083256d8.jpg'] \n\n# List of corrupted images found at dataset discussions\n\ncorrupted_images_present = segmentation_df[segmentation_df['ImageId'].isin(corrupted_images)]\n\n# As we see only one corrupted image is present\ncorrupted_images_present","metadata":{"_uuid":"e24f9606-723d-4447-9400-50b468182aa9","_cell_guid":"f3df625a-ea71-4302-9d33-c91eb0b44fb3","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.086831Z","iopub.execute_input":"2023-11-22T13:21:28.087246Z","iopub.status.idle":"2023-11-22T13:21:28.126678Z","shell.execute_reply.started":"2023-11-22T13:21:28.08721Z","shell.execute_reply":"2023-11-22T13:21:28.125602Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop corrupted\nsegmentation_df = segmentation_df.drop(corrupted_images_present.index) \nsegmentation_df[segmentation_df['ImageId'].isin(corrupted_images)]","metadata":{"_uuid":"bb13ffd0-17c9-45bc-88da-e1128031c89e","_cell_guid":"78f73538-2f0a-400c-a590-482d57d6ca3f","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.127963Z","iopub.execute_input":"2023-11-22T13:21:28.128321Z","iopub.status.idle":"2023-11-22T13:21:28.177714Z","shell.execute_reply.started":"2023-11-22T13:21:28.128287Z","shell.execute_reply":"2023-11-22T13:21:28.176257Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_combined_visualizations(ships_numbers):\n    fig, ax = plt.subplots(1, 2, figsize=(16, 8))\n\n    # Histogram\n    ax[0].hist(ships_numbers, bins=15)\n    ax[0].set_title('Distribution of number of ships')\n\n    # Pie chart\n    percentages = (ships_count.value_counts() / ships_count.value_counts().sum()) * 100\n    \n    labels = ['{0} - {1:1.2f} %'.format(i, j) for i, j in zip(ships_numbers.value_counts().index, percentages)]\n    \n    ax[1].pie(ships_numbers.value_counts(), labels=None)\n    ax[1].legend(labels, bbox_to_anchor=(1., 1.), fontsize=14)\n    ax[1].yaxis.set_visible(False)\n    ax[1].set_title('Distribution number of ships')\n\n    plt.show()","metadata":{"_uuid":"96802ef3-87fd-446a-ab8d-eb3307c77663","_cell_guid":"7e914fc4-1d63-4b1c-945e-3e94c8d6e7b5","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.179541Z","iopub.execute_input":"2023-11-22T13:21:28.180159Z","iopub.status.idle":"2023-11-22T13:21:28.189604Z","shell.execute_reply.started":"2023-11-22T13:21:28.180099Z","shell.execute_reply":"2023-11-22T13:21:28.188291Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Distribution of number of ships on image\nships_count = segmentation_df['ImageId'].value_counts()\nships_count[segmentation_df['ImageId'].loc[segmentation_df['EncodedPixels'].isna()]] = 0\nships_count.value_counts()","metadata":{"_uuid":"cc53d80d-e6d9-471e-9639-8c65088ee01a","_cell_guid":"4bbf2eae-e67e-419a-857d-406e27a65091","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.191373Z","iopub.execute_input":"2023-11-22T13:21:28.19192Z","iopub.status.idle":"2023-11-22T13:21:28.498835Z","shell.execute_reply.started":"2023-11-22T13:21:28.191887Z","shell.execute_reply":"2023-11-22T13:21:28.497717Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_combined_visualizations(ships_count)\n# As we see dataset is higly unbalanced, most images contains no ships - and so nearly no \n# information usefull to train segmentation model. Thats why it will usefull to balance this images.","metadata":{"_uuid":"dc06896a-8ad5-46f9-9d34-4b35fa4e464c","_cell_guid":"80b323e8-7159-4129-b5de-1f64b057e908","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:28.503422Z","iopub.execute_input":"2023-11-22T13:21:28.503809Z","iopub.status.idle":"2023-11-22T13:21:29.335186Z","shell.execute_reply.started":"2023-11-22T13:21:28.503775Z","shell.execute_reply":"2023-11-22T13:21:29.333828Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Balancing dataset by reducing images with zero ships to 10% of initial number\nrows_with_nan = segmentation_df[segmentation_df['EncodedPixels'].isnull()]\nrandom_sample_indexes = rows_with_nan.sample(frac=0.99).index\nbalanced_segmentation_df = segmentation_df.drop(random_sample_indexes)","metadata":{"_uuid":"2fbdb859-3fea-476b-aa80-d2d2fe91e22c","_cell_guid":"25af78c5-e82f-481a-92b9-de4547d10bb8","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:29.336526Z","iopub.execute_input":"2023-11-22T13:21:29.336854Z","iopub.status.idle":"2023-11-22T13:21:29.405131Z","shell.execute_reply.started":"2023-11-22T13:21:29.336825Z","shell.execute_reply":"2023-11-22T13:21:29.404297Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"balanced_segmentation_df","metadata":{"_uuid":"2e36a64c-c5e7-497a-8e5f-2720fd14db97","_cell_guid":"a1187b03-afa3-4a63-9d72-f68c220412b9","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:29.406214Z","iopub.execute_input":"2023-11-22T13:21:29.406498Z","iopub.status.idle":"2023-11-22T13:21:29.418176Z","shell.execute_reply.started":"2023-11-22T13:21:29.406473Z","shell.execute_reply":"2023-11-22T13:21:29.417315Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ships_count = balanced_segmentation_df['ImageId'].value_counts()\nships_count[balanced_segmentation_df['ImageId'].loc[balanced_segmentation_df['EncodedPixels'].isna()]] = 0\nships_count.value_counts()","metadata":{"_uuid":"291f3477-48f1-4a7d-8ab8-bb6332d7dd16","_cell_guid":"7e90b438-cde5-48f7-a6d9-5638543ccd88","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:29.419437Z","iopub.execute_input":"2023-11-22T13:21:29.419738Z","iopub.status.idle":"2023-11-22T13:21:29.490952Z","shell.execute_reply.started":"2023-11-22T13:21:29.419686Z","shell.execute_reply":"2023-11-22T13:21:29.490047Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"f3e947d3-0564-499e-b57b-3153ebe1a361","_cell_guid":"c3fa9d4e-320a-4380-b346-fdc46ae0632d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_combined_visualizations(ships_count)","metadata":{"_uuid":"fe9be040-3b85-425c-847f-84d1a88db21d","_cell_guid":"d003bcee-ef63-48b9-9fdb-3f2bc331bf4d","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:29.492238Z","iopub.execute_input":"2023-11-22T13:21:29.49253Z","iopub.status.idle":"2023-11-22T13:21:30.054146Z","shell.execute_reply.started":"2023-11-22T13:21:29.492505Z","shell.execute_reply":"2023-11-22T13:21:30.053223Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#finding ship area percentage on image\n\ndef pixels_percentage(encoded_pixels, shape):\n    \n    if pd.isnull(encoded_pixels):\n        return 0\n    \n    pixels_number = np.array(encoded_pixels.split()[1::2], dtype=int).sum()\n    \n    return pixels_number / (shape[0] * shape[1])\n\nbalanced_segmentation_df['ShipAreaPercentage'] = balanced_segmentation_df['EncodedPixels'].apply(lambda x: pixels_percentage(x, shape))","metadata":{"_uuid":"c38e746e-0c45-4e80-8e13-91faeeae7304","_cell_guid":"bbe3d8a4-06ee-402f-b464-af7edc63f730","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:30.055614Z","iopub.execute_input":"2023-11-22T13:21:30.056464Z","iopub.status.idle":"2023-11-22T13:21:31.588758Z","shell.execute_reply.started":"2023-11-22T13:21:30.056427Z","shell.execute_reply":"2023-11-22T13:21:31.587921Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('mean:', balanced_segmentation_df['ShipAreaPercentage'].mean())\nprint('std:', balanced_segmentation_df['ShipAreaPercentage'].std())\nprint('max:', balanced_segmentation_df['ShipAreaPercentage'].max())\n\n# Individual ship area reaches max 4.4% from total image area, so scaling images to lower resolutions can drastically impact perfomance","metadata":{"_uuid":"e3f26251-0b3d-4d8a-88aa-f32f2a9e0fc6","_cell_guid":"02184774-1480-40ba-8be3-7e33323e6ffd","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:31.589908Z","iopub.execute_input":"2023-11-22T13:21:31.590246Z","iopub.status.idle":"2023-11-22T13:21:31.598593Z","shell.execute_reply.started":"2023-11-22T13:21:31.590219Z","shell.execute_reply":"2023-11-22T13:21:31.597518Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_mask(image_id, shape):\n    shape = (768, 768)\n    rows = segmentation_df[segmentation_df['ImageId'] == image_id]\n    \n    ships_count = len(rows)\n    mask = np.zeros((shape[0] * shape[1]), dtype=np.uint8)\n    \n    for i in range(ships_count):\n        encoded_mask = rows.iloc[i]['EncodedPixels']\n        \n        if isinstance(encoded_mask, float):  # Check for NaN values\n            return mask.reshape((shape[0], shape[1], 1))\n        encoded_mask = np.array(encoded_mask.split(), dtype=int)\n        \n        rle_data = np.reshape(encoded_mask, (-1, 2))\n        \n        for pixel, shift in rle_data:\n            mask[pixel-1:pixel+shift] = 255\n            \n    \n    return mask.reshape((1, shape[1], shape[0])).T","metadata":{"_uuid":"13f5a206-522d-4ba6-bc27-7f32c6cd5e56","_cell_guid":"65394b25-093d-4dd7-9c8f-61507bae5027","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:31.599989Z","iopub.execute_input":"2023-11-22T13:21:31.6003Z","iopub.status.idle":"2023-11-22T13:21:31.610049Z","shell.execute_reply.started":"2023-11-22T13:21:31.600274Z","shell.execute_reply":"2023-11-22T13:21:31.608842Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path1 = balanced_segmentation_df['ImageId'][29]\npath2 = balanced_segmentation_df['ImageId'][4]\npath3 = balanced_segmentation_df['ImageId'][238]\n\nimage1 = cv2.imread(TRAIN_PATH + path1)\nimage2 = cv2.imread(TRAIN_PATH + path2)\nimage3 = cv2.imread(TRAIN_PATH + path3)\n\nmask1 = get_mask(path1, shape)\nmask2 = get_mask(path2, shape)\nmask3 = get_mask(path3, shape)\n\nf,ax = plt.subplots(3, 2, figsize=(10, 5 * 3))\nax[0][0].imshow(image1)\nax[0][1].imshow(mask1)\nax[1][0].imshow(image2)\nax[1][1].imshow(mask2)\nax[2][0].imshow(image3)\nax[2][1].imshow(mask3)\nplt.show()","metadata":{"_uuid":"1017d0bb-2b58-4fa2-b058-957b59885b25","_cell_guid":"04ebd4a8-0a0c-4ec0-a573-0e750000c8fa","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:31.611383Z","iopub.execute_input":"2023-11-22T13:21:31.611673Z","iopub.status.idle":"2023-11-22T13:21:33.276391Z","shell.execute_reply.started":"2023-11-22T13:21:31.611648Z","shell.execute_reply":"2023-11-22T13:21:33.275467Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"e1f553f1-0aad-4235-9701-c01eae89b95c","_cell_guid":"2549d8e1-b058-4cb9-83da-2f62525851d6","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nshape = (256, 256)\nbalanced_segmentation_df['ShipsCount'] = balanced_segmentation_df.groupby('ImageId')['ImageId'].transform('count')\nbalanced_segmentation_df.loc[balanced_segmentation_df['EncodedPixels'].isnull(), 'ShipsCount'] = 0\ntrain_df = balanced_segmentation_df[['ImageId', 'ShipsCount']].drop_duplicates()\n# train_df = train_df[0:5000]\ntrain_data, val_data = train_test_split(list(train_df['ImageId']), \n                                        test_size=0.05, \n                                        stratify=train_df['ShipsCount'], \n                                        random_state=42)","metadata":{"_uuid":"07856675-bdcf-4485-b8e9-8cc2c610b1d7","_cell_guid":"2f21cd98-40cb-43ca-81e6-68f669349e17","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:33.277656Z","iopub.execute_input":"2023-11-22T13:21:33.278Z","iopub.status.idle":"2023-11-22T13:21:33.983752Z","shell.execute_reply.started":"2023-11-22T13:21:33.277955Z","shell.execute_reply":"2023-11-22T13:21:33.982801Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train data size:', len(train_data))\nprint('Validtion data size:', len(val_data))","metadata":{"_uuid":"85bb190d-ad38-45c5-a520-2662619ceb2a","_cell_guid":"72c538cc-fe3b-41a5-b9f3-165ac8370315","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:33.984805Z","iopub.execute_input":"2023-11-22T13:21:33.985099Z","iopub.status.idle":"2023-11-22T13:21:33.990126Z","shell.execute_reply.started":"2023-11-22T13:21:33.985073Z","shell.execute_reply":"2023-11-22T13:21:33.989196Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing Utilities\nimport tensorflow as tf\n\ndef random_flip(input_image, input_mask):\n    if tf.random.uniform(()) > 0.5:\n        input_image = tf.image.flip_left_right(input_image)\n        input_mask = tf.image.flip_left_right(input_mask)\n\n    return input_image, input_mask\n\n\ndef normalize(input_image, input_mask):\n\n    input_image = tf.cast(input_image, tf.float32) / 255.0\n    input_mask = tf.cast(input_mask, tf.float32) / 255.0\n    return input_image, input_mask\n\n\n@tf.function\ndef load_image_train(image_name, shape):\n    '''resizes, normalizes, and flips the training data'''\n    input_image = tf.io.read_file(TRAIN_PATH + image_name)\n    input_image = tf.image.decode_png(input_image, channels=3)\n    input_image = tf.image.resize(input_image, shape, method='nearest')\n    input_mask = get_mask(image_name, shape)\n\n    input_mask = tf.image.resize(input_mask, shape, method='nearest')\n    \n    input_image, input_mask = random_flip(input_image, input_mask)\n    input_image, input_mask = normalize(input_image, input_mask)\n\n    return input_image, input_mask\n\n\ndef load_image_val(image_name, shape):\n    '''resizes and normalizes the test data'''\n    input_image = tf.io.read_file(TRAIN_PATH + image_name)\n    input_image = tf.image.decode_png(input_image, channels=3)\n    input_image = tf.image.resize(input_image, shape, method='nearest')\n    input_mask = get_mask(image_name, shape)\n    input_mask = tf.image.resize(input_mask, shape, method='nearest')\n    input_image, input_mask = normalize(input_image, input_mask)\n\n    return input_image, input_mask","metadata":{"_uuid":"5f8a75e3-44db-4e94-8bb6-d5d201ce9c7f","_cell_guid":"ca69d133-f49f-4274-94a3-8017a4390a18","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:33.991414Z","iopub.execute_input":"2023-11-22T13:21:33.991687Z","iopub.status.idle":"2023-11-22T13:21:45.453494Z","shell.execute_reply.started":"2023-11-22T13:21:33.991664Z","shell.execute_reply":"2023-11-22T13:21:45.452633Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef train_generator():\n    for image_id in train_data:\n        yield load_image_train(image_id, shape)\n        \ndef validation_generator():\n    for image_id in val_data:\n        yield load_image_val(image_id, shape)\n        \ntrain_dataset = tf.data.Dataset.from_generator(train_generator, \n                                               output_signature=(tf.TensorSpec(shape=(shape[0], shape[1], 3), dtype=tf.float32),\n                                                                 tf.TensorSpec(shape=(shape[0], shape[1], 1), dtype=tf.float32)))\nvalidation_dataset = tf.data.Dataset.from_generator(validation_generator,\n                                                    output_signature=(tf.TensorSpec(shape=(shape[0], shape[1], 3), dtype=tf.float32),\n                                                                 tf.TensorSpec(shape=(shape[0], shape[1], 1), dtype=tf.float32)))\n\n# val = train_data['ImageId'].map(lambda x: load_image_val(x, shape))","metadata":{"_uuid":"f7db3f74-803d-4404-8a87-73c1ea3c89e6","_cell_guid":"9da6bbcb-668a-4e08-89f8-40e516e5f22e","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:45.454786Z","iopub.execute_input":"2023-11-22T13:21:45.455528Z","iopub.status.idle":"2023-11-22T13:21:48.302447Z","shell.execute_reply.started":"2023-11-22T13:21:45.455492Z","shell.execute_reply":"2023-11-22T13:21:48.301467Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"_uuid":"f33ba786-2a15-45de-a0fb-4d10785d7fc0","_cell_guid":"89d48986-fe09-4417-a95f-12c291fd1399","trusted":true}},{"cell_type":"code","source":"for element in train_dataset.take(1):  # Taking one element for demonstration\n    for tensor in element:\n        print((tensor).shape)","metadata":{"_uuid":"468cb66f-c055-4337-9c2f-b9a23e23aa79","_cell_guid":"198ad063-0035-45fb-910b-d39bff5ea750","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:48.303715Z","iopub.execute_input":"2023-11-22T13:21:48.304069Z","iopub.status.idle":"2023-11-22T13:21:48.942951Z","shell.execute_reply.started":"2023-11-22T13:21:48.30404Z","shell.execute_reply":"2023-11-22T13:21:48.941802Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display(display_list,titles=[], display_string=None):\n  '''displays a list of images/masks'''\n\n  plt.figure(figsize=(15, 15))\n\n  for i in range(len(display_list)):\n    plt.subplot(1, len(display_list), i+1)\n    plt.title(titles[i])\n    plt.xticks([])\n    plt.yticks([])\n    if display_string and i == 1:\n      plt.xlabel(display_string, fontsize=12)\n    img_arr = tf.keras.preprocessing.image.array_to_img(display_list[i])\n    plt.imshow(img_arr)\n  \n  plt.show()\n\n\ndef show_image_from_dataset(dataset):\n  '''displays the first image and its mask from a dataset'''\n\n  for image, mask in dataset.take(23):\n    sample_image, sample_mask = image, mask\n\n  display([sample_image, sample_mask], titles=[\"Image\", \"True Mask\"])\n\nshow_image_from_dataset(train_dataset)\nshow_image_from_dataset(validation_dataset)","metadata":{"_uuid":"cf3c779f-ee21-455a-b9b7-4c4900edf190","_cell_guid":"d16e7843-2531-4413-97cd-28ffef0f2896","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:48.944586Z","iopub.execute_input":"2023-11-22T13:21:48.945006Z","iopub.status.idle":"2023-11-22T13:21:53.799847Z","shell.execute_reply.started":"2023-11-22T13:21:48.944946Z","shell.execute_reply":"2023-11-22T13:21:53.798689Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16\nBUFFER_SIZE = 1000\n\n# shuffle and group the train set into batches\n# train_dataset = train_dataset.cache().shuffle(BUFFER_SIZE).batch(BATCH_SIZE) # can overload ram so disabled in kaggle\ntrain_dataset = train_dataset.shuffle(BUFFER_SIZE).batch(BATCH_SIZE)\n# do a prefetch to optimize processing\n# train_dataset = train_dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)  # can overload ram so disabled in kaggle\n\nvalidation_dataset = validation_dataset.batch(BATCH_SIZE)","metadata":{"_uuid":"9b7fabc7-ecec-411f-87f0-526f1dfbab0b","_cell_guid":"e759ff02-65d0-4fc0-bd49-ae1ac7ae0e52","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:53.805736Z","iopub.execute_input":"2023-11-22T13:21:53.806398Z","iopub.status.idle":"2023-11-22T13:21:53.820635Z","shell.execute_reply.started":"2023-11-22T13:21:53.806358Z","shell.execute_reply":"2023-11-22T13:21:53.819743Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoder Utilities\n\ndef conv2d_block(input_tensor, n_filters, kernel_size = 3):\n  '''\n  Adds 2 convolutional layers with the parameters passed to it\n\n  Args:\n    input_tensor (tensor) -- the input tensor\n    n_filters (int) -- number of filters\n    kernel_size (int) -- kernel size for the convolution\n\n  Returns:\n    tensor of output features\n  '''\n  # first layer\n  x = input_tensor\n  for i in range(2):\n    x = tf.keras.layers.Conv2D(filters = n_filters, kernel_size = (kernel_size, kernel_size),\\\n            kernel_initializer = 'he_normal', padding = 'same')(x)\n    x = tf.keras.layers.Activation('relu')(x)\n  \n  return x\n\n\ndef encoder_block(inputs, n_filters=64, pool_size=(2,2), dropout=0.3):\n  '''\n  Adds two convolutional blocks and then perform down sampling on output of convolutions.\n\n  Args:\n    input_tensor (tensor) -- the input tensor\n    n_filters (int) -- number of filters\n    kernel_size (int) -- kernel size for the convolution\n\n  Returns:\n    f - the output features of the convolution block \n    p - the maxpooled features with dropout\n  '''\n\n  f = conv2d_block(inputs, n_filters=n_filters)\n  p = tf.keras.layers.MaxPooling2D(pool_size=(2,2))(f)\n  p = tf.keras.layers.Dropout(0.2)(p)\n\n  return f, p\n\n\ndef encoder(inputs):\n  '''\n  This function defines the encoder or downsampling path.\n\n  Args:\n    inputs (tensor) -- batch of input images\n\n  Returns:\n    p4 - the output maxpooled features of the last encoder block\n    (f1, f2, f3, f4) - the output features of all the encoder blocks\n  '''\n  f1, p1 = encoder_block(inputs, n_filters=8, pool_size=(2,2), dropout=0.3)\n  f2, p2 = encoder_block(p1, n_filters=16, pool_size=(2,2), dropout=0.3)\n  f3, p3 = encoder_block(p2, n_filters=32, pool_size=(2,2), dropout=0.3)\n  f4, p4 = encoder_block(p3, n_filters=64, pool_size=(2,2), dropout=0.3)\n\n  return p4, (f1, f2, f3, f4)\n\ndef bottleneck(inputs):\n  '''\n  This function defines the bottleneck convolutions to extract more features before the upsampling layers.\n  '''\n  \n  bottle_neck = conv2d_block(inputs, n_filters=128)\n\n  return bottle_neck\n\n# Decoder Utilities\n\ndef decoder_block(inputs, conv_output, n_filters=64, kernel_size=3, strides=3, dropout=0.2):\n  '''\n  defines the one decoder block of the UNet\n\n  Args:\n    inputs (tensor) -- batch of input features\n    conv_output (tensor) -- features from an encoder block\n    n_filters (int) -- number of filters\n    kernel_size (int) -- kernel size\n    strides (int) -- strides for the deconvolution/upsampling\n    padding (string) -- \"same\" or \"valid\", tells if shape will be preserved by zero padding\n\n  Returns:\n    c (tensor) -- output features of the decoder block\n  '''\n  u = tf.keras.layers.Conv2DTranspose(n_filters, kernel_size, strides = strides, padding = 'same')(inputs)\n  c = tf.keras.layers.concatenate([u, conv_output])\n  c = tf.keras.layers.Dropout(dropout)(c)\n  c = conv2d_block(c, n_filters, kernel_size=3)\n\n  return c\n\n\ndef decoder(inputs, convs, output_channels):\n  '''\n  Defines the decoder of the UNet chaining together 4 decoder blocks. \n  \n  Args:\n    inputs (tensor) -- batch of input features\n    convs (tuple) -- features from the encoder blocks\n    output_channels (int) -- number of classes in the label map\n\n  Returns:\n    outputs (tensor) -- the pixel wise label map of the image\n  '''\n  \n  f1, f2, f3, f4 = convs\n\n  c6 = decoder_block(inputs, f4, n_filters=64, kernel_size=(3,3), strides=(2,2), dropout=0.3)\n  c7 = decoder_block(c6, f3, n_filters=32, kernel_size=(3,3), strides=(2,2), dropout=0.3)\n  c8 = decoder_block(c7, f2, n_filters=16, kernel_size=(3,3), strides=(2,2), dropout=0.3)\n  c9 = decoder_block(c8, f1, n_filters=8, kernel_size=(3,3), strides=(2,2), dropout=0.3)\n\n  outputs = tf.keras.layers.Conv2D(output_channels, (1, 1), activation='sigmoid')(c9)\n\n  return outputs\n\nOUTPUT_CHANNELS = 1\n\ndef unet(shape):\n  '''\n  Defines the UNet by connecting the encoder, bottleneck and decoder.\n  '''\n\n  # specify the input shape\n  inputs = tf.keras.layers.Input(shape=(shape[0],shape[1], 3,))\n\n  # feed the inputs to the encoder\n  encoder_output, convs = encoder(inputs)\n\n  # feed the encoder output to the bottleneck\n  bottle_neck = bottleneck(encoder_output)\n\n  # feed the bottleneck and encoder block outputs to the decoder\n  # specify the number of classes via the `output_channels` argument\n  outputs = decoder(bottle_neck, convs, output_channels=OUTPUT_CHANNELS)\n  \n  # create the model\n  model = tf.keras.Model(inputs=inputs, outputs=outputs)\n\n  return model\n\n# instantiate the model\nmodel = unet(shape)\n\n# see the resulting model architecture\nmodel.summary()","metadata":{"_uuid":"5021ace1-a00e-4e76-a8d6-04ee0d19ff5b","_cell_guid":"5bce962d-e804-4332-8df2-caa0387ce5a2","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:53.822471Z","iopub.execute_input":"2023-11-22T13:21:53.822861Z","iopub.status.idle":"2023-11-22T13:21:54.642407Z","shell.execute_reply.started":"2023-11-22T13:21:53.822826Z","shell.execute_reply":"2023-11-22T13:21:54.641529Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"15f85b57-aa1b-4306-9d76-d0b268db699c","_cell_guid":"81c99636-0f21-42e9-b949-44c14f90dd90","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"798aef00-e654-459a-b907-9cc99ab50813","_cell_guid":"92eca8bc-d0d8-41d2-8aeb-9165841fb2ac","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"_uuid":"6e1f805c-2986-407a-bc6d-1c04490289c6","_cell_guid":"c7a5a67e-a033-47ad-b7fd-8d3ad8feebd8","trusted":true}},{"cell_type":"code","source":"import tensorflow as tf\n\n\ndef dice_coeff(y_true, y_pred, smooth=1e-5):\n    intersection = tf.reduce_sum(y_true * y_pred, axis=(1,2,3))\n    sum_of_squares_pred = tf.reduce_sum(y_pred, axis=(1,2,3))\n    sum_of_squares_true = tf.reduce_sum(y_true, axis=(1,2,3))\n    dice = (2. * intersection + smooth) / (sum_of_squares_pred + sum_of_squares_true + smooth)\n    return dice\n\ndef combined_loss(y_true, y_pred):\n    alpha = 0.5  # Weight for BCE loss\n    bce_loss = tf.keras.losses.BinaryCrossentropy()(y_true, y_pred)\n    dice = dice_coeff(y_true, y_pred)\n    combined = alpha * bce_loss - tf.math.log(1. - dice)  # Combine BCE and Dice loss\n    return combined\n\n# def bce_dice_loss(y_true, y_pred):\n#     return 1e-3*tf.keras.losses.BinaryCrossentropy()(y_true, y_pred) + dice_loss(y_true, y_pred)\n\n# def dice_coeff(y_true, y_pred):\n#     smooth = 1e-15\n#     intersection = tf.reduce_sum(y_true * y_pred)\n#     union = tf.reduce_sum(y_true) + tf.reduce_sum(y_pred)\n#     dice = (2. * intersection + smooth) / (union + smooth)\n#     return dice\n\n# Define combined loss function\n\n\n# def true_positive_rate(y_true, y_pred):\n#     return K.sum(K.flatten(y_true)*K.flatten(K.round(y_pred)))/K.sum(y_true)\n\n# def dice_coef(y_true, y_pred):\n#     y_true_f = K.flatten(y_true)\n#     y_pred = K.cast(y_pred, 'float32')\n#     y_pred_f = K.cast(K.greater(K.flatten(y_pred), 0.5), 'float32')\n#     intersection = y_true_f * y_pred_f\n#     score = 2. * K.sum(intersection) / (K.sum(y_true_f) + K.sum(y_pred_f))\n#     return score\n\n# def bce_logdice_loss(y_true, y_pred):\n#     return binary_crossentropy(y_true, y_pred) - K.log(1. - dice_loss(y_true, y_pred))\n\n# def dice_loss(y_true, y_pred):\n#     smooth = 1.\n#     y_true_f = K.flatten(y_true)\n#     y_pred_f = K.flatten(y_pred)\n#     intersection = y_true_f * y_pred_f\n#     score = (2. * K.sum(intersection) + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n#     return 1. - score\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(1e-4), loss=combined_loss, metrics=[dice_coeff, 'binary_accuracy', true_positive_rate])","metadata":{"_uuid":"688528c5-6d97-4ffb-8a48-59cd26937314","_cell_guid":"0c38529b-fef8-436f-9dda-f74345fb5775","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:54.643674Z","iopub.execute_input":"2023-11-22T13:21:54.644614Z","iopub.status.idle":"2023-11-22T13:21:54.672325Z","shell.execute_reply.started":"2023-11-22T13:21:54.644583Z","shell.execute_reply":"2023-11-22T13:21:54.671317Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # configure the training parameters and train the model\n\n# EPOCHS = 10\n\n# # this will take around 20 minutes to run\n# model_history = model.fit(train_dataset, epochs=EPOCHS, validation_data=validation_dataset)","metadata":{"_uuid":"78d7401a-5c9f-4463-bd2a-0a9e862f3678","_cell_guid":"d25273d0-5e85-4db1-ba29-8652e40c9b52","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:54.673651Z","iopub.execute_input":"2023-11-22T13:21:54.673963Z","iopub.status.idle":"2023-11-22T13:21:54.683329Z","shell.execute_reply.started":"2023-11-22T13:21:54.673927Z","shell.execute_reply":"2023-11-22T13:21:54.682293Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_filepath = '/kaggle/working/checkpoints/model-checkpoint'\n\nsave_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    mode='max',\n    save_best_only=True\n)","metadata":{"_uuid":"55ef51bd-420c-421b-9d15-a0d0ac090a94","_cell_guid":"fde93f64-2d44-47dd-921c-81faba8e02b3","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:54.684498Z","iopub.execute_input":"2023-11-22T13:21:54.68478Z","iopub.status.idle":"2023-11-22T13:21:54.696524Z","shell.execute_reply.started":"2023-11-22T13:21:54.684756Z","shell.execute_reply":"2023-11-22T13:21:54.6956Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"_uuid":"a0d42f15-591e-4afd-ab33-78bc6dcdaf6d","_cell_guid":"673c5680-9c0e-48b4-b8f4-cdca2d4531c7","trusted":true}},{"cell_type":"code","source":"EPOCHS = 5\n\nmodel.fit(train_dataset,\n                          epochs=EPOCHS,\n                          steps_per_epoch=200,\n                          validation_data=validation_dataset,\n                          callbacks=[save_callback])","metadata":{"_uuid":"4fb23186-2d3a-437f-8f0d-a99b6918d726","_cell_guid":"d9c4cc63-d566-4260-8eac-40d4923f86bc","collapsed":false,"execution":{"iopub.status.busy":"2023-11-22T13:21:54.697915Z","iopub.execute_input":"2023-11-22T13:21:54.698268Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EPOCHS = 5\n# STEPS_PER_EPOCH = len(train_data) // BATCH_SIZE\n\n# optimizer = tfa.optimizers.RectifiedAdam(\n#     learning_rate=0.005,\n#     total_steps=EPOCHS * STEPS_PER_EPOCH,\n#     warmup_proportion=0.3,\n#     min_lr=0.00001,\n# )\n\n\n# optimizer = tfa.optimizers.Lookahead(optimizer)\n\n# model = UNetModel(shape + (3,)).model\n\n# model.compile(optimizer='Adam', \n#               loss=bce_logdice_loss, # bce_dice_loss,\n#               metrics=[dice_coef, 'binary_accuracy', true_positive_rate],)\n\n# model.fit(train_dataset,\n#                           epochs=EPOCHS,\n                          \n#                           validation_data=validation_dataset,)","metadata":{"_uuid":"aa593d85-e69e-4ed7-b953-ce276e7f1850","_cell_guid":"c151b92e-3e41-4813-adcf-265f04da24c6","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"02b36d11-d222-4ceb-ab24-ed0cd8e02752","_cell_guid":"bdfd4c5b-5aac-4bad-834c-f3e53d8d49cf","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"54b33256-799e-4bed-ad6a-924b74e6b909","_cell_guid":"7537176f-5272-483e-985e-74809c927276","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"931efbe7-19f3-4243-a7f4-68c2af0e7940","_cell_guid":"2149faf9-fe28-4b3e-b04e-2ce2a1a962fd","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"f620b356-3b36-4313-811c-2adbd4075c13","_cell_guid":"1cd9832f-850d-426f-ae07-5bab5f6a41d2","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}