{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# External downloads (will need to be downloaded apart, packaged as a dataset in Kaggle an manually imported for the submission notebook)\n\nINTERNET = False\n\nif INTERNET:\n    # --- Internet ON: ---\n    ! pip install segmentation_models     #! pip download segmentation_models -d ./\n    \nelse:\n    # --- Internet OFF: ---\n    ! pip install segmentation_models -f ../input/efficientnetb5weights --no-index\n\n\n# Pretrained Model weights:\n# https://github.com/Callidior/keras-applications/releases/download/efficientnet/efficientnet-b5_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5\n","metadata":{"execution":{"iopub.status.busy":"2022-10-11T11:05:28.689248Z","iopub.execute_input":"2022-10-11T11:05:28.690128Z","iopub.status.idle":"2022-10-11T11:05:41.003032Z","shell.execute_reply.started":"2022-10-11T11:05:28.690038Z","shell.execute_reply":"2022-10-11T11:05:41.001875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport keras\nfrom keras import utils as np_utils\nfrom pathlib import Path\nimport albumentations as A\nfrom time import time\nimport random\n\nimport segmentation_models as sm\nfrom segmentation_models import Unet\n\nrandom.seed(42)\nnp.random.seed(42)\ntf.random.set_seed(42)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T11:20:46.558786Z","iopub.execute_input":"2022-10-11T11:20:46.559212Z","iopub.status.idle":"2022-10-11T11:20:52.616573Z","shell.execute_reply.started":"2022-10-11T11:20:46.559173Z","shell.execute_reply":"2022-10-11T11:20:52.615530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_PATH = Path(\"../input/hubmap-organ-segmentation\")\nTRAIN_IMAGES_PATH = INPUT_PATH / \"train_images\"\nTRAIN_CSV_PATH = INPUT_PATH / \"train.csv\"\nTEST_IMAGES_PATH = INPUT_PATH / \"test_images\"\nTEST_CSV_PATH = INPUT_PATH / \"test.csv\"\nWEIGHTS_PATH = \"../input/efficientnetb5weights/efficientnet-b5_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5\"","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:13:01.942517Z","iopub.execute_input":"2022-10-11T08:13:01.944420Z","iopub.status.idle":"2022-10-11T08:13:01.949869Z","shell.execute_reply.started":"2022-10-11T08:13:01.944381Z","shell.execute_reply":"2022-10-11T08:13:01.948871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# UTILS\n#https://www.kaggle.com/code/pestipeti/decoding-rle-masks/notebook\ndef mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    \n    return \" \".join(str(x) for x in runs)\n\n\ndef rle2mask(mask_rle, shape=(3000,3000)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0::2], s[1::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n        \n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:13:01.953731Z","iopub.execute_input":"2022-10-11T08:13:01.954149Z","iopub.status.idle":"2022-10-11T08:13:01.966468Z","shell.execute_reply.started":"2022-10-11T08:13:01.954110Z","shell.execute_reply":"2022-10-11T08:13:01.965509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Augmentation\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\n\nimg_id = 11890\n_image = tf.keras.utils.load_img(TRAIN_IMAGES_PATH / f\"{img_id}.tiff\", target_size=(3000, 3000))\n_image = tf.keras.utils.img_to_array(_image, dtype=\"uint8\")\n\n\n_mask = rle2mask(train_df.loc[train_df.id == img_id].rle.values[0]).reshape((3000,3000,1))\n\nplt.rcParams[\"figure.figsize\"] = (7, 7)\nplt.imshow(_image)\nplt.imshow(_mask, alpha=0.4, cmap='gray')\nplt.axis(\"off\")\nplt.show()\n\n\ndef transform_f(img_size):\n    \n    transform = A.Compose([\n        A.OneOf(\n            [\n                #A.Affine(scale=0.8, mode=2, p=0.4),\n                #A.CenterCrop(height=500, width=500, p=1),\n                A.CenterCrop(height=2300, width=2300, p=1),\n                A.CropNonEmptyMaskIfExists(height=2500, width=2500, p=1),\n                A.CropNonEmptyMaskIfExists(height=2000, width=2000, p=1),\n                A.CropNonEmptyMaskIfExists(height=1500, width=1500, p=1),\n            ],\n            p=0.7\n        ),\n        A.Resize(img_size, img_size),\n        #A.ElasticTransform(p=0.95),\n        #A.ShiftScaleRotate(p=0.95),\n        A.PiecewiseAffine(scale=(0.005, 0.015), p=0.5),\n        A.Rotate(limit=90, p=0.8),\n        A.Flip(p=0.7),\n        #A.RandomCrop(width=2000, height=2000),\n        #A.RandomSizedCrop(min_max_height=(), height=3000, widht=3000),\n        #A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1, p=0.5),\n        A.ColorJitter(brightness=0.1, contrast=0.1, saturation=0.2, hue=0.1, p=0.8)\n    ])\n    \n    return transform\n\n\ntransform = transform_f(800)\n\ngrid_size = 3\nfig, ax = plt.subplots(nrows=grid_size, ncols=grid_size, figsize=(15, 15))\n\nfor i in range(grid_size):\n    for j in range(grid_size):\n\n        transformed = transform(image=_image, mask=_mask)\n\n        transformed_image = transformed[\"image\"]\n        transformed_mask = transformed[\"mask\"]\n\n        print(\"sizes:\", transformed_image.shape, transformed_mask.shape)\n        #ax[i][j].set_title(\"sizes:\", transformed_image.shape, transformed_mask.shape)\n        ax[i][j].imshow(transformed[\"image\"])\n        ax[i][j].imshow(transformed[\"mask\"], alpha=0.3, cmap='gray')\n        ax[i][j].axis(\"off\")\n    \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:13:01.968000Z","iopub.execute_input":"2022-10-11T08:13:01.968356Z","iopub.status.idle":"2022-10-11T08:13:15.676393Z","shell.execute_reply.started":"2022-10-11T08:13:01.968324Z","shell.execute_reply":"2022-10-11T08:13:15.675505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ./DataSequence.py\n\nsm.set_framework(\"tf.keras\")\nsm.framework()\n\n\n# SliceSequence resizes each image to img_Resize and takes slices as stored by the script Dfslices in the Dataframe\nclass SliceSequence(tf.keras.utils.Sequence):\n\n    def __init__(self, batch_size, img_size, df, img_Resize = 0, shuffle = True):\n        super().__init__()\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.df = df\n        self.img_Resize = img_Resize\n        self.indexes = np.arange(0, int(np.floor(len(df) / batch_size)) * batch_size)\n        if(shuffle):\n            np.random.shuffle(self.indexes)\n\n\n    def __len__(self):\n        return int(np.floor(len(self.df) / self.batch_size))\n\n    def __getitem__(self, idx):\n        ind = self.indexes[idx : idx+self.batch_size]\n\n        x = np.zeros((self.batch_size, self.img_size, self.img_size, 3), dtype=\"uint8\")\n        y = np.zeros((self.batch_size, self.img_size, self.img_size, 1), dtype=\"uint8\")\n\n        IdPrev = -1\n        for j, (indexm, row) in enumerate(self.df.iloc[ind].iterrows()):\n            if(IdPrev != int(row[\"id\"])):\n                img_path = TRAIN_IMAGES_PATH / (str(int(row[\"id\"])) + \".tiff\")\n                if(self.img_Resize > 0):\n                    img = tf.keras.utils.load_img(img_path, target_size=(self.img_Resize, self.img_Resize))\n                else:\n                    img = tf.keras.utils.load_img(img_path)\n                img_array = tf.keras.utils.img_to_array(img, dtype=\"uint8\")\n            Xslice = int(row[\"Xslice\"])\n            Yslice = int(row[\"Yslice\"])\n            x[j] = img_array[Xslice : Xslice+self.img_size, Yslice : Yslice+self.img_size]\n\n            # calculate mask\n            if (IdPrev != int(row[\"id\"])):\n                w = int(row[\"img_width\"])\n                h = int(row[\"img_height\"])\n                rle = row[\"rle\"]\n                s = rle.split()\n                starts, lengths = [np.asarray(t, dtype=\"int\") for t in (s[0:][::2], s[1:][::2])]\n                starts = starts - 1\n                original_mask = np.zeros(h * w, dtype=np.uint8)\n                for s, l in zip(starts, lengths):\n                    original_mask[s:s + l] = 1\n                original_mask = original_mask.reshape((h, w)).T\n                original_mask = tf.keras.utils.array_to_img(original_mask[:, :, tf.newaxis], scale=False)\n\n                # resize mask\n                if (self.img_Resize > 0):\n                    original_mask = original_mask.resize((self.img_Resize, self.img_Resize))\n                mask_array = tf.keras.utils.img_to_array(original_mask, dtype=\"uint8\")\n            y[j] = mask_array[Xslice:Xslice+self.img_size, Yslice:Yslice+self.img_size]\n            IdPrev = int(row[\"id\"])\n            \n        return x.astype(\"float32\") / 255, y\n    \n    \n## ImSequence with a resize of each image to img_size\nclass ImSequence(tf.keras.utils.Sequence):\n\n    def __init__(self, batch_size, img_size, df, transform=None, set_type=\"train\"):\n        super().__init__()\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.df = df\n        self.set_type = set_type\n        self.img_path = TRAIN_IMAGES_PATH if (self.set_type == \"train\" or self.set_type == \"val_test\") else TEST_IMAGES_PATH\n        self.transform = transform\n\n        \n    def __len__(self):\n        return int(np.floor(len(self.df) / self.batch_size))\n\n    \n    def __getitem__(self, idx):\n        ind = np.arange(idx * self.batch_size, (idx + 1) * self.batch_size) % len(self.df)\n\n        x = np.zeros((self.batch_size, self.img_size, self.img_size, 3), dtype=\"uint8\")\n        y = np.zeros((self.batch_size, self.img_size, self.img_size, 1), dtype=\"uint8\")\n\n        for j, (indexm, row) in enumerate(self.df.iloc[ind].iterrows()):\n            img_path = self.img_path / (str(row[\"id\"]) + \".tiff\")\n            #img = tf.keras.utils.load_img(img_path, target_size=(self.img_size, self.img_size))\n\n            # calculate mask\n            if self.set_type == \"train\":\n                \n                img = tf.keras.utils.load_img(img_path, target_size=(3000, 3000))\n                img_array = tf.keras.utils.img_to_array(img, dtype=\"uint8\")\n                \n                w = row[\"img_width\"]\n                h = row[\"img_height\"]\n                rle = row[\"rle\"]\n                original_mask = rle2mask(rle)\n                # we need at least three channels to save an image so here expand mask to (3000, 3000, 1)\n                original_mask = tf.keras.utils.array_to_img(original_mask[:, :, tf.newaxis], scale=False)\n\n                # resize mask\n                #mask = original_mask.resize((self.img_size, self.img_size))\n                #mask_array = tf.keras.utils.img_to_array(mask, dtype=\"uint8\").reshape((self.img_size,self.img_size, 1))\n                mask = original_mask.resize((3000, 3000))\n                mask_array = tf.keras.utils.img_to_array(mask, dtype=\"uint8\").reshape((3000, 3000, 1))\n                \n                # data augmentation                                    \n                transformed = self.transform(image=img_array, mask=mask_array)\n                transformed_image = transformed[\"image\"]\n                transformed_mask = transformed[\"mask\"]\n                \n                x[j] = transformed_image\n                y[j] = transformed_mask\n                \n            else:\n                img = tf.keras.utils.load_img(img_path, target_size=(self.img_size, self.img_size))\n                img_array = tf.keras.utils.img_to_array(img, dtype=\"uint8\")\n                \n                if self.transform:\n                    x = np.zeros((len(self.transform)+1, self.img_size, self.img_size, 3), dtype=\"uint8\")\n                    x[0] = img_array\n                    \n                    for j, transform_i in enumerate(self.transform):\n                        transformed = transform_i(image=img_array)\n                        transformed_image = transformed[\"image\"]\n                        x[j+1] = transformed_image\n                    \n                else:\n                    x[j] = img_array \n\n        if self.set_type == \"train\":\n            return x.astype(\"float32\") / 255, y\n            \n        else:\n            return x.astype(\"float32\") / 255","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:46:13.178465Z","iopub.execute_input":"2022-10-11T08:46:13.178828Z","iopub.status.idle":"2022-10-11T08:46:13.209679Z","shell.execute_reply.started":"2022-10-11T08:46:13.178795Z","shell.execute_reply":"2022-10-11T08:46:13.208432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ./ModelTrain.py\n\ndf_test = pd.read_csv(TEST_CSV_PATH)\n\n# --- hparams ---\n#LR = 0.0001\nBACKBONE = \"efficientnetb5\"\nBATCH_SIZE = 4\nIMG_SIZE = 640\nEPOCHS = 50 if len(df_test) > 16 else 1  # treak to avoid spending gpu time again on notebook saving run, then it will run fine at submission\nVAL_SET_SIZE = 0.05\n\n\nif INTERNET:\n    # Internet ON:\n    model = Unet(backbone_name=BACKBONE, encoder_weights=\"imagenet\", encoder_freeze=True)\n\nelse:\n    # Internet OFF:\n    model = Unet(backbone_name=BACKBONE, encoder_weights=WEIGHTS_PATH, encoder_freeze=True)\n\n# Segmentation models losses can be combined together by '+' and scaled by integer or float factor\ndice_loss = sm.losses.DiceLoss()\n\n# actulally total_loss can be imported directly from library, above example just show you how to manipulate with losses\n# total_loss = sm.losses.binary_focal_dice_loss # or sm.losses.categorical_focal_dice_loss \n\n# compile keras model with defined optimozer, loss and metrics\nmodel.compile(\"Adam\", \"binary_crossentropy\", [sm.metrics.FScore()])\n\n\ndf = pd.read_csv(TRAIN_CSV_PATH)\n\n# Filter the data related to the organ with which is going to be tested\n\n# Number of samples taken for validation\n\nNVal = int(np.floor(len(df) * VAL_SET_SIZE))\n\ndfshuffle = df.iloc[np.random.permutation(len(df))]\n\ndv = dfshuffle.iloc[-NVal:]\ndt = dfshuffle.iloc[:-NVal]\n\ntransform = transform_f(IMG_SIZE)\n\n## ImSequence with a resize of each image to img_size\n\nTrain = ImSequence(BATCH_SIZE, IMG_SIZE, dt, transform)\nVal = ImSequence(BATCH_SIZE, IMG_SIZE, dv, transform)\n\n\n## SliceSequence resizes each image to img_Resize and takes slices as stored by the script Dfslices in the Dataframe\n\n#Train = SliceSequence(batch_size, img_size, dt, img_Resize=img_Resize,  shuffle=False)\n#Val = SliceSequence(batch_size, img_size, dv, img_Resize=img_Resize, shuffle=False)\n\n\ncallbacks = [\n    keras.callbacks.ModelCheckpoint('./best_model.h5', save_weights_only=True, save_best_only=True, mode='min'),\n    keras.callbacks.ReduceLROnPlateau(),\n]\n\nhistory = model.fit(Train, epochs=EPOCHS, callbacks=callbacks, validation_data=Val)\n\n\n# score plots\nepochs = range(EPOCHS)\nfig, (ax0, ax1, ax2) = plt.subplots(nrows=1, ncols=3, sharex=True, figsize=(20, 6))\nax0.set_title(\"Loss\")\nax0.plot(epochs, history.history[\"loss\"])\nax0.plot(epochs, history.history[\"val_loss\"])\nax1.set_title(\"f1-score\")\nax1.plot(epochs, history.history[\"f1-score\"])\nax1.plot(epochs, history.history[\"val_f1-score\"])\nax2.set_title(\"lr\")\nax2.plot(epochs, history.history[\"lr\"])\nfig.suptitle(f\"{BACKBONE} | batch_size={BATCH_SIZE} | img_size={IMG_SIZE}\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:13:15.727813Z","iopub.execute_input":"2022-10-11T08:13:15.728410Z","iopub.status.idle":"2022-10-11T08:18:21.043678Z","shell.execute_reply.started":"2022-10-11T08:13:15.728372Z","shell.execute_reply":"2022-10-11T08:18:21.042776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nProvisional val f1-score result logs:\n\n- EffB7, batch_size=16, img_size=256, epochs=15      --> 0.56 aprox\n- resnet18, batch_size=16, img_size=256, epochs=15   --> 0.1 (and worse) BUG\n- resnet34, batch_size=8, img_size=256, epochs=15    --> 0.1 (and worse) BUG\n- EffB6, batch_size=8, img_size=256, epochs=15       --> 0.65 \n- EffB5, batch_size=4, img_size=512, epochs=15       --> 0.71 / TEST: 0.43\n\nAfter applying DA in train and val:           (now the validation score is more similar to the competition test data) \n- EffB5, batch_size=4, img_size=512, epochs=25                                 --> 0.58 / TEST: 0.47 \n- EffB5, batch_size=4, img_size=512, epochs=50                                 --> 0.67 / TEST: 0.43\n- EffB3, batch_size=4, img_size=800, epochs=30, DA large+small crops/resizes   --> TEST: 0.46\n\n\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-10-11T08:18:21.044959Z","iopub.execute_input":"2022-10-11T08:18:21.045935Z","iopub.status.idle":"2022-10-11T08:18:21.054383Z","shell.execute_reply.started":"2022-10-11T08:18:21.045896Z","shell.execute_reply":"2022-10-11T08:18:21.052866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Time Augmentation (tta)\n\n# tta functions\ntta_transforms = [\n    A.HorizontalFlip(p=1),\n    A.VerticalFlip(p=1),\n    A.Transpose(p=1),\n    A.Compose([\n        A.Transpose(p=1),\n        A.VerticalFlip(p=1),\n    ])\n]\n\ntta_inv_transforms = [\n    A.HorizontalFlip(p=1),\n    A.VerticalFlip(p=1),\n    A.Transpose(p=1),\n    A.Compose([\n        A.VerticalFlip(p=1),\n        A.Transpose(p=1),\n    ])\n]\n    \n# Viz, test and debug\nfig, ax = plt.subplots(nrows=4, ncols=len(tta_transforms), figsize=(15, 15))\n\nax[0][0].imshow(_image)\nax[0][0].imshow(_mask, alpha=0.3, cmap='gray')\nfor i in range(len(tta_transforms)):\n    ax[0][i].axis(\"off\")\n\nfor j, transform in zip(range(len(tta_transforms)), tta_transforms):\n\n    transformed = transform(image=_image, mask=_mask)\n\n    transformed_image = transformed[\"image\"]\n    transformed_mask = transformed[\"mask\"]\n\n    ax[1][j].imshow(transformed_image)\n    ax[1][j].imshow(transformed_mask, alpha=0.3, cmap='gray')\n    ax[1][j].axis(\"off\")\n    \n    ax[2][j].imshow(transformed_mask, alpha=0.3, cmap='gray')\n    ax[2][j].axis(\"off\")\n    \n    # inverse\n    transformed = tta_inv_transforms[j](image=np.concatenate((transformed_mask, transformed_mask, transformed_mask), axis=2), mask=transformed_mask)\n\n    #transformed_image = transformed[\"image\"]\n    transformed_mask = transformed[\"mask\"]\n\n    #ax[2][j].imshow(transformed_image)\n    ax[3][j].imshow(transformed_mask, alpha=0.3, cmap='gray')\n    ax[3][j].axis(\"off\")\n     \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T09:09:34.773578Z","iopub.execute_input":"2022-10-11T09:09:34.774222Z","iopub.status.idle":"2022-10-11T09:09:53.212360Z","shell.execute_reply.started":"2022-10-11T09:09:34.774188Z","shell.execute_reply":"2022-10-11T09:09:53.210782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation inference for debugging and demo\nmodel.load_weights(\"../input/hubmap-trained-model/best_model.h5\")\n\nTEST_BATCH_SIZE = 1   # with tta transforms, the current logic just supports batches of one image\n\ndf_submission = pd.DataFrame(columns=[\"id\", \"rle\"]).set_index(\"id\")\n\ndf_test = dv.copy().reset_index()\n\n# for debugging\nadd_id_debug = lambda : 0\nif len(df_test) < 10:\n    add_id_debug = lambda : np.random.randint(100)\n    for i, o in zip(range(1,5), [\"prostate\", \"lung\", \"kidney\", \"largeintestine\"]):\n        df_test.loc[i] = df_test.loc[0]\n        df_test.loc[i, \"organ\"] = o\n        \n#df_test.loc[1] = df_test.iloc[0]  # DELETE debugging with 2 copies of the test image\nremaining_test = len(df_test)\n\ntest_i = 0\n\nwhile remaining_test > 0:\n    batch_size = TEST_BATCH_SIZE if remaining_test >= TEST_BATCH_SIZE else remaining_test\n    test_imgs = ImSequence(batch_size, IMG_SIZE, df_test.iloc[test_i : test_i+TEST_BATCH_SIZE], transform=tta_transforms, set_type=\"val_test\")\n    \n    test_masks = model.predict(test_imgs)  # taking out round as we need the raw float values to first inv_transform, then avg and finally rounded in 0s and 1s\n    \n    test_id = df_test.iloc[test_i : test_i+TEST_BATCH_SIZE][\"id\"].values.tolist()[0]\n    organ = df_test.iloc[test_i : test_i+TEST_BATCH_SIZE][\"organ\"].values.tolist()[0]\n    \n    for i, test_mask, inv_transform in zip(range(len(test_masks)), test_masks, [None]+tta_inv_transforms):\n        if not inv_transform:\n            continue\n        else:\n            inv_transformed = inv_transform(image=np.concatenate((test_mask, test_mask, test_mask), axis=2), mask=test_mask)\n            test_masks[i] = inv_transformed[\"mask\"]\n            \n    test_mask = test_masks.mean(axis=0).round()\n            \n    size = df_test.loc[df_test[\"id\"] == test_id][[\"img_height\", \"img_width\"]].values\n    test_mask_i = test_mask.copy()\n\n    test_mask_i = tf.keras.utils.array_to_img(test_mask_i, scale=False)\n\n    # resize mask\n    test_mask_i = test_mask_i.resize((size[0][0], size[0][1]))\n    test_mask_i = tf.keras.utils.img_to_array(test_mask_i, dtype=\"uint8\")\n    rescaled_mask = test_mask_i.reshape((size[0][0], size[0][1], 1))\n\n    test_rle = mask2rle(rescaled_mask)\n    df_submission.loc[test_id+add_id_debug(), :] = [test_rle]\n\n    # plot results vs ground truth\n    fig, ax = plt.subplots(nrows=1, ncols=2, figsize=(10, 5))\n\n    ax[0].imshow(test_imgs[0][0])\n    ax[0].imshow(test_mask, alpha=0.5, cmap='gray')\n    ax[0].axis(\"off\")\n    ax[0].set_title(\"Predicted mask\")\n    \n    gt_mask = rle2mask(train_df.loc[train_df.id == test_id].rle.values[0]).reshape((3000,3000,1))\n    \n    transform_resize = A.Resize(IMG_SIZE, IMG_SIZE)\n    resized = transform_resize(image=_image, mask=gt_mask)\n    resized_gt_mask = resized[\"mask\"]\n    \n    ax[1].imshow(test_imgs[0][0])\n    ax[1].imshow(resized_gt_mask, alpha=0.5, cmap='gray')\n    ax[1].axis(\"off\")\n    ax[1].set_title(\"Ground truth\")\n\n    fig.suptitle(f\"val img id: {test_id}  |  organ: {organ}\", fontsize=14)\n    fig.show()\n        \n        \n    remaining_test -= TEST_BATCH_SIZE\n    test_i += TEST_BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2022-10-11T09:17:27.322752Z","iopub.execute_input":"2022-10-11T09:17:27.323148Z","iopub.status.idle":"2022-10-11T09:17:50.372682Z","shell.execute_reply.started":"2022-10-11T09:17:27.323116Z","shell.execute_reply":"2022-10-11T09:17:50.371782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction and file submission\nmodel.load_weights(\"../input/hubmap-trained-model/best_model.h5\")\n\nTEST_BATCH_SIZE = 1   # with tta transforms, the current logic just supports batches of one image\n\ndf_submission = pd.DataFrame(columns=[\"id\", \"rle\"]).set_index(\"id\")\n\ndf_test = pd.read_csv(TEST_CSV_PATH)\n\n# for debugging\nadd_id_debug = lambda : 0\nif len(df_test) < 10:\n    add_id_debug = lambda : np.random.randint(100)\n    for i, o in zip(range(1,5), [\"prostate\", \"lung\", \"kidney\", \"largeintestine\"]):\n        df_test.loc[i] = df_test.loc[0]\n        df_test.loc[i, \"organ\"] = o\n        \n#df_test.loc[1] = df_test.iloc[0]  # DELETE debugging with 2 copies of the test image\nremaining_test = len(df_test)\n\ntest_i = 0\n\nwhile remaining_test > 0:\n    batch_size = TEST_BATCH_SIZE if remaining_test >= TEST_BATCH_SIZE else remaining_test\n    test_imgs = ImSequence(batch_size, IMG_SIZE, df_test.iloc[test_i : test_i+TEST_BATCH_SIZE], transform=tta_transforms, set_type=\"test\")\n    \n    test_masks = model.predict(test_imgs)  # taking out round as we need the raw float values to first inv_transform, then avg and finally rounded in 0s and 1s\n    \n    test_id = df_test.iloc[test_i : test_i+TEST_BATCH_SIZE][\"id\"].values.tolist()[0]\n    \n    for i, test_mask, inv_transform in zip(range(len(test_masks)), test_masks, [None]+tta_inv_transforms):\n        if not inv_transform:\n            continue\n        else:\n            inv_transformed = inv_transform(image=np.concatenate((test_mask, test_mask, test_mask), axis=2), mask=test_mask)\n            test_masks[i] = inv_transformed[\"mask\"]\n            \n    test_mask = test_masks.mean(axis=0).round()\n            \n    size = df_test.loc[df_test[\"id\"] == test_id][[\"img_height\", \"img_width\"]].values\n    test_mask_i = test_mask.copy()\n\n    test_mask_i = tf.keras.utils.array_to_img(test_mask_i, scale=False)\n\n    # resize mask\n    test_mask_i = test_mask_i.resize((size[0][0], size[0][1]))\n    test_mask_i = tf.keras.utils.img_to_array(test_mask_i, dtype=\"uint8\")\n    rescaled_mask = test_mask_i.reshape((size[0][0], size[0][1], 1))\n\n    test_rle = mask2rle(rescaled_mask)\n    df_submission.loc[test_id+add_id_debug(), :] = [test_rle]\n\n    if len(df_test) < 20:\n        plt.imshow(test_imgs[0][0])\n        plt.imshow(test_mask, alpha=0.5, cmap='gray')\n        plt.axis(\"off\")\n        plt.show()\n        \n    remaining_test -= TEST_BATCH_SIZE\n    test_i += TEST_BATCH_SIZE","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2022-09-04T13:06:43.706155Z","iopub.execute_input":"2022-09-04T13:06:43.706529Z","iopub.status.idle":"2022-09-04T13:06:43.728856Z","shell.execute_reply.started":"2022-09-04T13:06:43.706493Z","shell.execute_reply":"2022-09-04T13:06:43.727654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-09-01T11:04:06.343069Z","iopub.status.idle":"2022-09-01T11:04:06.344112Z","shell.execute_reply.started":"2022-09-01T11:04:06.343842Z","shell.execute_reply":"2022-09-01T11:04:06.343867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- Next steps: --- \n\n# - Adapt ImSequence to properly handle all test images  [DONE]\n# - Resize back output mask from prediction, then transform to rle  [DONE]\n# - Fill submission file with all rles from test images  [DONE]\n\n# - Better data augmentation pipeline with Albumentations (follow example in github)  [DONE]\n\n# - Train and predict with different models for each organ  [DONE (similar or worse results)]\n# - Sliced dataset with proper sizes  \n# - Research other backbones a part from efficientnet (ResNet was performing bad)\n# - Try to improve dataloaders\n# - Research about loss function and also model callbacks and checkpoints  [checkpoints DONE]\n# - Implement model checkpointing to avoid long waitings for results (except for the final best submissions which should be entirely trained)  [DONE but not allowed in competition]\n# - Copy other public code solutions with good scores","metadata":{"execution":{"iopub.status.busy":"2022-09-01T11:04:06.345523Z","iopub.status.idle":"2022-09-01T11:04:06.346603Z","shell.execute_reply.started":"2022-09-01T11:04:06.346328Z","shell.execute_reply":"2022-09-01T11:04:06.346352Z"},"trusted":true},"execution_count":null,"outputs":[]}]}