{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center> Kaggle: HuBMAP + HPA - Hacking the Human Body </center>\n\nThe goal of this competition is to identify the locations of each functional tissue unit (FTU) in biopsy slides from several different organs. The underlying data includes imagery from different sources prepared with different protocols at a variety of resolutions, reflecting typical challenges for working with medical data.","metadata":{}},{"cell_type":"markdown","source":"<p id=\"table\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #BF2C58; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#imports\">1&nbsp;&nbsp;&nbsp;&nbsp;INSTALLS & IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#RLE\">2&nbsp;&nbsp;&nbsp;&nbsp;RLE CONVERTORS AND PLOTTING</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#EDA\">3&nbsp;&nbsp;&nbsp;&nbsp;EDA</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#dataset\">4&nbsp;&nbsp;&nbsp;&nbsp;DATASET</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#dataset_entry\">5&nbsp;&nbsp;&nbsp;&nbsp;DATASET ENTRYPOINT</a></h3>\n\n---","metadata":{}},{"cell_type":"markdown","source":"<a id=\"imports\"></a>\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"imports\">1&nbsp;&nbsp;INSTALLS & IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#table\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"!pip install nptyping","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:24:11.346917Z","iopub.execute_input":"2022-09-19T18:24:11.347310Z","iopub.status.idle":"2022-09-19T18:24:26.230014Z","shell.execute_reply.started":"2022-09-19T18:24:11.347256Z","shell.execute_reply":"2022-09-19T18:24:26.228712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport os\nimport shutil\nimport warnings\nfrom typing import Callable, Optional, Tuple, Union\n\nimport albumentations as A\nimport cv2\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport plotly.graph_objects as go\nimport seaborn as sns\nimport tifffile as tiff\nimport torch\nfrom nptyping import NDArray, Float64, UInt8, Shape\nfrom sklearn.model_selection import train_test_split\nfrom torch.nn import Module\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\n\nsns.set_style(\"darkgrid\")\nwarnings.filterwarnings('ignore')\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:24:30.498334Z","iopub.execute_input":"2022-09-19T18:24:30.498715Z","iopub.status.idle":"2022-09-19T18:24:33.931203Z","shell.execute_reply.started":"2022-09-19T18:24:30.498683Z","shell.execute_reply":"2022-09-19T18:24:33.930071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"RLE\"></a>\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"rle\">2&nbsp;&nbsp;RLE CONVERTORS AND PLOTTING&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#table\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"def mask2rle(img: NDArray[Shape['*, *'], UInt8]) -> str:\n    \"\"\"\n    Parameters\n    ----------\n    img : NDArray[Shape['*, *'], UInt8]\n        Binary image mask.\n    Returns\n    -------\n    str\n        Running length encoded string.\n    \"\"\"\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef rle2mask(mask_rle: str, shape: Tuple[int, int]) -> NDArray[Shape['*, *'], UInt8]:\n    \"\"\"\n    Parameters\n    ----------\n    mask_rle : str\n        String representation of running length encoded.\n    shape : Tuple[int, int]\n        Size of the image in the (height, width) format.\n    Returns\n    -------\n    NDArray[Shape['*, *'], UInt8]\n        Binary mask with shape according to the {shape} variable.\n    \"\"\"\n    mask_str = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (mask_str[0:][::2], mask_str[1:][::2])]\n    ends = lengths + starts\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for low, high in zip(starts, ends):\n        img[low:high] = 1\n    return img.reshape(shape).T\n\n\ndef show_image_from_dataset(img_id: int,\n                            df: pd.DataFrame,\n                            path: str = '../input/hubmap-organ-segmentation/train_images/',\n                            is_title: Optional[str] = None,\n                            figsize: Tuple[int, int] = (10, 5)) -> matplotlib.figure.Figure:\n    \"\"\"\n    Parameters\n    ----------\n    img_id : int\n        Image id from the ['id'] column.\n    df : pd.DataFrame\n        Dataframe, containing image ['id'] and ['rle'] columns.\n    path : str\n        Path to the image folder.\n    is_title : Optional[str]\n        Make title for the plotted image.\n    figsize : Tuple[int, int]\n        Shape of the output image in the (width, height) format.\n    Returns\n    -------\n    matplotlib.figure.Figure\n        Plotting the image.\n    \"\"\"\n    img = tiff.imread(path + str(img_id) + \".tiff\")\n    height, width = img.shape[0], img.shape[1]\n    mask = rle2mask(df[df[\"id\"] == img_id][\"rle\"].iloc[-1], (height, width))\n    plt.figure(figsize=(10, 8));\n    plt.imshow(img, zorder=1);\n    plt.imshow(mask, cmap='coolwarm', alpha=0.5, zorder=1);\n    if is_title:\n        plt.title(is_title, fontsize=15, pad=20)\n    \n\ndef show_image_with_mask(img:  NDArray[Shape['*, *, 3'], Float64],\n                         mask:  NDArray[Shape['*, *'], UInt8]) -> matplotlib.figure.Figure:\n    \"\"\"\n    Parameters\n    ----------\n    img : NDArray[Shape['*, *, *'], Float64]\n        Image as a n-dimensional numpy array.\n    mask : NDArray[Shape['*, *'], UInt8]\n        Binary mask as a n-dimensional numpy array.\n    Returns\n    -------\n    matplotlib.figure.Figure\n        Plotting the image with its mask.\n    \"\"\"\n    plt.figure(figsize=(10, 8));\n    plt.imshow(img, zorder=1);\n    plt.imshow(mask, cmap='coolwarm', alpha=0.5, zorder=1);","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:24:33.933088Z","iopub.execute_input":"2022-09-19T18:24:33.933623Z","iopub.status.idle":"2022-09-19T18:24:33.955457Z","shell.execute_reply.started":"2022-09-19T18:24:33.933592Z","shell.execute_reply":"2022-09-19T18:24:33.954093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"EDA\"></a>\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"eda\">3&nbsp;&nbsp;EDA&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#table\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\ntest_df = pd.read_csv('../input/hubmap-organ-segmentation/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:24:34.410724Z","iopub.execute_input":"2022-09-19T18:24:34.411408Z","iopub.status.idle":"2022-09-19T18:24:34.930510Z","shell.execute_reply.started":"2022-09-19T18:24:34.411355Z","shell.execute_reply":"2022-09-19T18:24:34.929356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:24:36.585658Z","iopub.execute_input":"2022-09-19T18:24:36.586775Z","iopub.status.idle":"2022-09-19T18:24:36.611729Z","shell.execute_reply.started":"2022-09-19T18:24:36.586727Z","shell.execute_reply":"2022-09-19T18:24:36.610765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dataset description:\n- *id* - The image ID.\n- *organ* - The organ that the biopsy sample was taken from.\n- *data_source* - Whether the image was provided by Hubamp or HPA.\n- *img_height* - The height of the image in pixels.\n- *img_width* - The width of the image in pixels.\n- *pixel_size* - The height/width of a single pixel from this image in micrometers.<br> All HPA images have a pixel size of 0.4 µm. For Hubmap imagery the pixel size is:<br>\n    - 0.5 µm for kidney\n    - 0.2290 µm for large intestine\n    - 0.7562 µm for lung\n    - 0.4945 µm for spleen\n    - 6.263 µm for prostate.\n- *tissue_thickness* - The thickness of the biopsy sample in micrometers.<br> All HPA images have a thickness of 4 µm. The Hubmap samples have tissue slice thicknesses:\n    - 10 µm for kidney\n    - 8 µm for large intestine\n    - 4 µm for spleen\n    - 5 µm for lung\n    - and 5 µm for prostate.\n- *rle* - The target column. A run length encoded copy of the annotations. **Provided for the training set only.**\n- *age* - The patient's age in years. **Provided for the training set only.**\n- *sex* - The sex of the patient. **Provided for the training set only.**","metadata":{}},{"cell_type":"markdown","source":"Lets have a closer look at the image with its segmented tissues","metadata":{}},{"cell_type":"code","source":"train_df[train_df['organ'] == 'prostate']['id'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:25:28.924938Z","iopub.execute_input":"2022-09-19T18:25:28.925685Z","iopub.status.idle":"2022-09-19T18:25:28.942022Z","shell.execute_reply.started":"2022-09-19T18:25:28.925644Z","shell.execute_reply":"2022-09-19T18:25:28.940763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image_from_dataset(10274, train_df)","metadata":{"execution":{"iopub.status.busy":"2022-09-19T18:25:55.090167Z","iopub.execute_input":"2022-09-19T18:25:55.090871Z","iopub.status.idle":"2022-09-19T18:25:57.926655Z","shell.execute_reply.started":"2022-09-19T18:25:55.090831Z","shell.execute_reply":"2022-09-19T18:25:57.925588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=1, ncols=2, figsize=(16,8));\nsns.histplot(train_df['img_height'], bins=50, color='red', ax=ax[0]);\nax[0].set_xlabel('Image height in pixels', fontsize=16);\nax[0].set_ylabel('Number of images', fontsize=16);\nax[0].set_title('Histogram of the training set image resolutions', fontsize=16, pad=20);\nsns.histplot(train_df['img_width'], bins=50, color='red', ax=ax[1]);\nax[1].set_xlabel('Image width in pixels', fontsize=16);\nax[1].set_ylabel('Number of images', fontsize=16);\nax[1].set_title('Histogram of the training set image width distribution', fontsize=16, pad=20);\nfig.suptitle('Distribution of training set image shape characteristics', fontsize=18, y=1);","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As it can be observerd, almost all images from the training set have 3000x3000 resolution, the floating range is [2308, 3070]. This have to be considered in choosing NN model architecture in general and type of transposed convolutions in particular.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,8));\nsns.violinplot(data=train_df, x='sex', y='age', palette='viridis');\nplt.xlabel('Sex', fontsize=15);\nplt.xticks(fontsize=14);\nplt.ylabel('Age', fontsize=15);\nplt.yticks(np.arange(10, 101, 5), fontsize=13);\nplt.title('Representation of patients ages divided by their sex', fontsize=16, pad=20);\nprint('Max female age:\\t {}'.format(train_df[train_df['sex'] == 'Female']['age'].max()), end='\\t')\nprint('Min female age:\\t {}'.format(train_df[train_df['sex'] == 'Female']['age'].min()))\nprint('Max male age:\\t {}'.format(train_df[train_df['sex'] == 'Male']['age'].max()), end='\\t')\nprint('Min male age:\\t {}'.format(train_df[train_df['sex'] == 'Male']['age'].min()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Distributions of age groups are very close one to each other except the higher amount of male patients between 55 and 60 years old. <br>\nFurther investigations are needed to say whether this feature is important or not.","metadata":{}},{"cell_type":"markdown","source":"Now, let's observe the distribution of the samples across differernt organs.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,8));\nsns.countplot(data=train_df,\n              x='organ',\n              orient='v',\n              palette='viridis',\n              order=train_df['organ'].value_counts(ascending=True).index);\nplt.title('Biopsy samples distribution across different organs', fontsize=16, pad=20);\nplt.xticks(fontsize=13);\nplt.xlabel('Organ', fontsize=14);\nplt.yticks(np.arange(0, 101, 5), fontsize=13);\nplt.ylabel('Number of samples', fontsize=14);","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Imballance can be observed here. The most common class (kidney) has 2 times more training samples than the rarest one (lung). In contrast to this, spleen and largeintestine classes have approximately the same number of data points.","metadata":{}},{"cell_type":"markdown","source":"Let's have a look not only on the amount of training data but also on the varition among segmented areas grouped by classes. To achieve it, it is needed to create a pivot table.","metadata":{}},{"cell_type":"code","source":"def pixel_precentage_counter(row: pd.Series) -> float:\n    \"\"\"\n    Parameters\n    ----------\n    row : pd.Series\n        Row from the training dataframe.\n    Returns\n    -------\n    float\n        How many precents of the image is covered by segmented area.\n    \"\"\"\n    img_height = row['img_height']\n    img_width = row['img_width']\n    rle = row['rle'].split()\n    pixel_amount = 0\n    for index, value in enumerate(rle):\n        if index % 2 != 0:\n            pixel_amount += int(value)\n    return 100 * pixel_amount / (img_height * img_width)\n\n\ntrain_df['segments_average'] = train_df.apply(pixel_precentage_counter, axis=1)\n\nsegments_precentage_pivot = pd.pivot_table(train_df,\n                                           index=['organ'],\n                                           values=['segments_average', 'age'],\n                                           aggfunc={'segments_average': np.mean,\n                                                    'age': 'count'}).reset_index()\nsegments_precentage_pivot.columns = ['organ', 'items_count', 'segments_average']\nsegments_precentage_pivot = segments_precentage_pivot.sort_values(by='segments_average')\nsegments_precentage_pivot","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(data=[go.Scatter(\n    x=segments_precentage_pivot['organ'].values,\n    y=segments_precentage_pivot['items_count'].values,\n    text=['organ: lung<br>samples: 48<br>avg segments area: 2.08%',\n          'organ: kidney<br>samples: 99<br>avg segments area: 3.02%',\n          'organ: spleen<br>samples: 53<br>avg segments area: 10.1%',\n          'organ: prostate<br>samples: 93<br>avg segments area: 15.4%',\n          'organ: largeintestine<br>samples: 58<br>avg segments area: 19.34%'],\n    mode='markers',\n    marker=dict(\n        colorscale='magma',\n        color=segments_precentage_pivot['segments_average'].values,\n        size=segments_precentage_pivot['segments_average'].values * 10,\n        colorbar=dict(title=\"Area, %\"),\n        showscale=True))])\n\nfig.update_layout(\n    title=\"Bubble plot representing average segmented area precentage according to different organs\",\n    xaxis_title=\"Organs\",\n    yaxis_title=\"Amount of samples\")\n\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lung and kidney organs have 2.08% and 3.02% of average segmented areas respectively. The most significant segmentation coverage is in the largeintestine class - 19.34% in average. Keeping this in mind, it is important to analyse pictures itself (e.g. using show_image_with_mask function) and, if the majority of organs will have various segment shapes, than, probably smart choice is to train not one but a couple of neural networks. Each of them will have to segment functional tissues in its particular organ or set of organs.","metadata":{}},{"cell_type":"markdown","source":"<a id=\"dataset\"></a>\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"dataset\">4&nbsp;&nbsp;DATASET&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#table\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"class SegmentationDataset(Dataset):\n    def __init__(self,\n                 df: pd.DataFrame,\n                 img_path: str,\n                 transform: Optional[A.Compose] = None,\n                 is_train: bool = True,\n                 is_valid: bool = False,\n                 device: str = 'cuda'):\n        self.df = df\n        self.img_path = img_path\n        self.transform = transform\n        self.is_train = is_train\n        self.is_valid = is_valid\n        self.device = device \n    \n    def __len__(self) -> int:\n        return len(self.df)\n        \n    \n    def __getitem__(self, idx: int) -> Union[Tuple[torch.Tensor, torch.Tensor],\n                                             Tuple[torch.Tensor]]:\n        row = self.df.iloc[idx]\n        img_id = row['id']\n        img = tiff.imread(self.img_path + str(img_id) + \".tiff\")\n        if self.is_train or self.is_valid:\n            rle = row['rle']\n            mask = rle2mask(rle, (img.shape[1], img.shape[0]))\n            if self.transform:\n                transformed = self.transform(image=img, mask=mask)\n                img = transformed['image']\n                mask = transformed['mask']\n                img = torch.Tensor(img)\n#                 print(img.shape)\n                img = torch.permute(img, (1, 2, 0)).to(self.device)\n#                 print(img.shape)\n#                 print('\\n\\n\\n')\n                mask = torch.Tensor(mask).to(self.device)\n#                 mask = torch.permute(mask, (2, 1, 0)).to(self.device)\n#                 print(mask.shape)\n            return img, mask\n        else:\n            if self.transform:\n                transformed = self.transform(image=img)\n                img = transformed['image']\n#                 img = torch.Tensor(img).unsqueeze(0)\n#                 img = torch.permute(img, (0, 3, 2, 1)).to(self.device)\n            return img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_cropper(img: Union[NDArray[Shape['*, *, 3'], Float64], NDArray[Shape['*, *'], UInt8]],\n                  height_min_border: int,\n                  height_max_border: Union[int, None],\n                  width_min_border: int,\n                  width_max_border: Union[int, None]) -> Union[NDArray[Shape['*, *, 3'], Float64],\n                                                               NDArray[Shape['*, *'], UInt8]]:\n    \"\"\"\n    Parameters\n    ----------\n    img: np.ndarray\n        Image that will be cropped.\n    height_max_border : Union[int, None]\n        Y-axis max border for image croppping.\n    height_min_border : int\n        Y-axis min border for image croppping.\n    width_max_border : Union[int, None]\n        X-axis max border for image croppping.\n    width_min_border : int\n        X-axis min border for image croppping.\n    Returns\n    -------\n    Union[NDArray[Shape['*, *, 3'], Float64], NDArray[Shape['*, *'], UInt8]]\n        Cropped image/mask according to defined parameters.\n    \"\"\"\n    sub_img = img[height_min_border:height_max_border,\n                  width_min_border:width_max_border]\n    return sub_img\n\n\ndef segmentation_decomposition(path_to_save: str,\n                               df_to_initial_data: pd.DataFrame,\n                               img_column_name: str,\n                               img_column_type: Callable,\n                               rle_mask_column_name: str,\n                               path_to_initial_data: str = '../input/hubmap-organ-segmentation/train_images/',\n                               is_train: bool = True,\n                               units: int = 4) -> pd.DataFrame:\n    \"\"\"\n    Parameters\n    ----------\n    path_to_save : str\n        Path for storing new dataset.\n    df_to_initial_data : pd.DataFrame\n        Dataframe, containing image filename and mask decoded running length.\n    path_to_initial_data: str\n        Path where original dataset is located.\n    is_train : bool\n        Defines whether new dataset id train or not. For train dataset\n        there is addditional creation of rle from binary mask.\n    units : int\n        The number of subimages from the original image.\n    Returns\n    -------\n    pd.DataFrame\n        Create a new dataset in '{path_to_save}' folder\n        with '{unit}' times more data points that was originally.\n        Creates and return new dataframe, containing renewed filenames\n        and running length encoded masks.\n    \"\"\"\n    new_data = pd.DataFrame(columns=[['id', 'rle']])\n    for file in tqdm(sorted(os.listdir(path_to_initial_data))):\n        file_without_extension = img_column_type(file.split('.')[0])\n        df_sample = df_to_initial_data[df_to_initial_data[img_column_name] == file_without_extension]\n        img = tiff.imread(path_to_initial_data + file)\n        if is_train:\n            mask_rle = df_sample[rle_mask_column_name].values[0]\n            mask = rle2mask(mask_rle, (img.shape[0], img.shape[1]))\n        blocks_segment = int(math.sqrt(units))\n        height_step = img.shape[0] // blocks_segment\n        width_step = img.shape[1] // blocks_segment\n        index_counter = 1\n        for unit_height in range(blocks_segment):\n            for unit_width in range(blocks_segment):\n                parameters_to_save = []\n                new_filename = str(file_without_extension) + f'_{index_counter}'\n                parameters_to_save.append(new_filename)\n                \n                height_min_border = height_step * unit_height\n                height_max_border = height_step * (unit_height + 1)\n                width_min_border = width_step * unit_width\n                width_max_border = width_step * (unit_width + 1)\n                \n                if unit_width == unit_height == blocks_segment - 1:\n                    height_max_border = None\n                    width_max_border = None\n                    sub_img = image_cropper(img, height_min_border, height_max_border, width_min_border, width_max_border)\n                    \n                elif unit_width == blocks_segment - 1:\n                    width_max_border = None\n                    sub_img = image_cropper(img, height_min_border, height_max_border, width_min_border, width_max_border)\n                    \n                elif unit_height == blocks_segment - 1:\n                    height_max_border = None\n                    sub_img = image_cropper(img, height_min_border, height_max_border, width_min_border, width_max_border)\n                \n                else:\n                    sub_img = image_cropper(img, height_min_border, height_max_border, width_min_border, width_max_border)\n            \n                if is_train:\n                    sub_mask = image_cropper(mask, height_min_border, height_max_border, width_min_border, width_max_border)\n                    sub_mask_rle = mask2rle(sub_mask)\n                    parameters_to_save.append(sub_mask_rle)\n                \n                tiff.imsave(path_to_save + new_filename + '.tiff', sub_img)\n                new_data.loc[len(new_data)] = parameters_to_save\n                index_counter += 1\n    return new_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_dataloader(df: pd.DataFrame,\n                    size: int,\n                    path: str,\n                    batch_size: int,\n                    is_train: bool = True,\n                    is_valid: bool = False,\n                    is_decomposition: bool = True,\n                    device: str = 'cuda') -> DataLoader:\n    \"\"\"\n    Parameters\n    ----------\n    df : pd.DataFrame\n        Dataframe with image ids for training.\n    size : int\n        Image resize size (both width and height).\n    path : str\n        Path to the image folder.\n    batch_size : int\n        The number of images in training loader batch.\n    is_train : bool\n        Defines whether this a training set or not.\n    device : str\n        Device to perform computation on (either 'cuda' or 'cpu').\n    Returns\n    -------\n    DataLoader\n        Train DataLoader with images and their masks.\n    \"\"\"\n    train_transform = A.Compose(\n        [\n            A.Resize(size, size),\n            A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.5),\n            A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ]\n    )\n    \n    test_transform = A.Compose(\n        [\n            A.Resize(size, size),\n            A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ]\n    )\n    \n    dataset = SegmentationDataset(df=df,\n                                  img_path=path,\n                                  transform=train_transform, # if (is_train or is_valid) else test_transform,\n                                  is_train=is_train,\n                                  is_valid=is_valid,\n                                  device=device)\n    \n    dataloader = DataLoader(dataset,\n                            batch_size=batch_size,\n                            shuffle=True if is_train else False,\n                            drop_last=True if is_valid else False)\n    return dataloader\n\n\ndef make_validation(df: pd.DataFrame,\n                    test_size: float = 0.1) -> Tuple[pd.DataFrame,\n                                                     pd.DataFrame]:\n    \"\"\"\n    Parameters\n    ----------\n    df : pd.DataFrame\n        Dataframe to split.\n    test_size : float\n        The precentage of validation part after\n        the dataset splitted. Ranges between 0 and 1.\n    Returns\n    -------\n    DataLoader\n        Train DataLoader with images and their masks.\n    \"\"\"\n    train, test = train_test_split(df,\n                                   test_size=test_size,\n                                   random_state=RANDOM_STATE)\n    return train, test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"dataset_entry\"></a>\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"dataset_entry\">5&nbsp;&nbsp;DATASET ENTRYPOINT&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#table\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"NEW_PATH = '../working/new_data_folder/'\nORIGINAL_PATH = '../input/hubmap-organ-segmentation/train_images/'\nUNITS = 4\nTRANSFORMS = False\nRESIZE = 560\nIS_DECOMPOSITION = True\nBS = 4\nRANDOM_STATE = 42\nDEVICE = 'cuda'\n\n\ndef dataset_entry(train_df: pd.DataFrame,\n                  valid_size: Optional[float] = None,\n                  add_valid: bool = False) -> Union[DataLoader,\n                                                    Tuple[DataLoader, DataLoader]]:\n    \"\"\"\n    Parameters\n    ----------\n    train_df : pd.DataFrame\n        Train dataframe to make dataloader from.\n    valid_size : Optional[float]\n        The amount of valid size in th dataset. Ranges from 0 to 1.\n    add_valid : bool\n        Parameter to regulate whether to add validation or not\n    Returns\n    -------\n    float\n        How many precents of the image is covered by segmented area.\n    \"\"\"\n    if IS_DECOMPOSITION:\n        !rm -rf $NEW_PATH\n        !mkdir $NEW_PATH\n        train_df = segmentation_decomposition(path_to_save=NEW_PATH,\n                                              df_to_initial_data=train_df,\n                                              img_column_name='id',\n                                              img_column_type=int,\n                                              rle_mask_column_name='rle',\n                                              path_to_initial_data=ORIGINAL_PATH,\n                                              is_train=True,\n                                              units=UNITS)\n        \n    path = ORIGINAL_PATH if not IS_DECOMPOSITION else NEW_PATH\n    \n    if add_valid:\n        train_df, valid_df = make_validation(train_df, valid_size)\n        valid_dataloader = make_dataloader(df=valid_df,\n                                           size=RESIZE,\n                                           path=path,\n                                           batch_size=BS,\n                                           is_train=False,\n                                           is_valid=True,\n                                           is_decomposition=IS_DECOMPOSITION,\n                                           device=DEVICE)\n    \n    train_dataloader = make_dataloader(df=train_df,\n                                       size=RESIZE,\n                                       path=path,\n                                       batch_size=BS,\n                                       is_train=True,\n                                       is_valid=False,\n                                       is_decomposition=IS_DECOMPOSITION,\n                                       device=DEVICE)\n\n    return train_dataloader, valid_dataloader if add_valid else None\n\n\ntrain_dataloader, valid_dataloader = dataset_entry(train_df=train_df, valid_size=0.2, add_valid=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}