{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HuBMAP_HPA Kaggle competition","metadata":{}},{"cell_type":"markdown","source":"## Preface","metadata":{}},{"cell_type":"markdown","source":"First of all, thank you to all members of the organisation for making this kaggle competition, as well as provide us with clear guidelines and quality data to work from.\nAlso, huge thanks to DARIEN SCHETTLER for his quality work on exploration which eased my start on data exploration (https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom tensorflow.keras import layers\nfrom tensorflow.data import Dataset\nfrom sklearn.model_selection import train_test_split\nfrom itertools import product","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:37.742076Z","iopub.execute_input":"2022-08-28T09:56:37.742529Z","iopub.status.idle":"2022-08-28T09:56:39.931768Z","shell.execute_reply.started":"2022-08-28T09:56:37.742493Z","shell.execute_reply":"2022-08-28T09:56:39.929876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:39.934220Z","iopub.execute_input":"2022-08-28T09:56:39.934621Z","iopub.status.idle":"2022-08-28T09:56:39.944702Z","shell.execute_reply.started":"2022-08-28T09:56:39.934586Z","shell.execute_reply":"2022-08-28T09:56:39.943269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input/hubmap-organ-segmentation/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:39.947519Z","iopub.execute_input":"2022-08-28T09:56:39.952719Z","iopub.status.idle":"2022-08-28T09:56:40.128711Z","shell.execute_reply.started":"2022-08-28T09:56:39.952660Z","shell.execute_reply":"2022-08-28T09:56:40.127437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Data","metadata":{}},{"cell_type":"code","source":"datadir = \"/kaggle/input/hubmap-organ-segmentation/\"\ntrain_df = pd.read_csv(datadir+\"train.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:40.131240Z","iopub.execute_input":"2022-08-28T09:56:40.131623Z","iopub.status.idle":"2022-08-28T09:56:40.521830Z","shell.execute_reply.started":"2022-08-28T09:56:40.131589Z","shell.execute_reply":"2022-08-28T09:56:40.520739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"# modified from: https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# modified from: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle: str, shape: tuple) -> np.array: \n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T\n\n# modified from: https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation\n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(mask: np.array) -> str:\n    \"\"\" \n    mask: numpy array, 1 - mask, 0 - background\n    returns run-length as string formated\n    \"\"\"\n    pixels = mask.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef plot_img_mask(data_train: pd.DataFrame, id_img: int = None, func_for_id: callable = None, col_for_func: str = None):\n    \"\"\"\n    Again, thanks to DARIEN SCHETTLER whithout whom it would have taken me a while to figure out cv2.addWeighted()\n    inputs:\n        data_train: pandas dataframe loaded from train.csv\n        id_img: id of image in said dataframe\n        OR\n        func_for_id & col_for_func: define function to filter dataframe, and column on which to apply filter.\n            exemple: func_for_id = min , col_for_func = \"img_height\" -> we retrieve id of image with minimum height\n    \"\"\"\n    datadir = \"/kaggle/input/hubmap-organ-segmentation\"\n    if id_img:\n        data_plot = data_train[data_train['id'] == id_img].squeeze()\n    else:\n        if func_for_id and col_for_func:\n            data_plot = data_train[data_train[col_for_func] == func_for_id(data_train[col_for_func])].squeeze()\n            id_img = data_plot['id'].squeeze()\n        else:\n            id_img = data_plot['id'].sample(n=1)\n    plt.figure(figsize=(6,6))\n    img = Image.open(f'{datadir}/train_images/{id_img}.tiff')\n    img = np.array(img).astype(np.float32) / 255\n    mask = rle_decode(data_plot['rle'], (data_plot['img_height'], data_plot['img_width'])).astype(np.float32)\n    new_mask = np.stack([mask,]*3, axis=-1)*(124, 252, 0) / 255\n    new_mask = new_mask.astype(np.float32)\n    overlay = cv2.addWeighted(src1=img, alpha=0.99, \n                              src2=new_mask, beta=0.2, \n                              gamma=0)\n    plt.title(f\"id: {id_img}\")\n    plt.imshow(overlay)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:30:43.881787Z","iopub.execute_input":"2022-08-28T10:30:43.882261Z","iopub.status.idle":"2022-08-28T10:30:43.922962Z","shell.execute_reply.started":"2022-08-28T10:30:43.882226Z","shell.execute_reply":"2022-08-28T10:30:43.921435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploration","metadata":{}},{"cell_type":"code","source":"train_df['pixel_size'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:40.545813Z","iopub.execute_input":"2022-08-28T09:56:40.547115Z","iopub.status.idle":"2022-08-28T09:56:40.570358Z","shell.execute_reply.started":"2022-08-28T09:56:40.547075Z","shell.execute_reply":"2022-08-28T09:56:40.569457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['img_height'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:40.571594Z","iopub.execute_input":"2022-08-28T09:56:40.573650Z","iopub.status.idle":"2022-08-28T09:56:40.584200Z","shell.execute_reply.started":"2022-08-28T09:56:40.573613Z","shell.execute_reply":"2022-08-28T09:56:40.583291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['img_width'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:40.585771Z","iopub.execute_input":"2022-08-28T09:56:40.586493Z","iopub.status.idle":"2022-08-28T09:56:40.596696Z","shell.execute_reply.started":"2022-08-28T09:56:40.586456Z","shell.execute_reply":"2022-08-28T09:56:40.595386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img_mask(train_df, func_for_id = min, col_for_func = \"img_height\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:40.598539Z","iopub.execute_input":"2022-08-28T09:56:40.599261Z","iopub.status.idle":"2022-08-28T09:56:43.068172Z","shell.execute_reply.started":"2022-08-28T09:56:40.599223Z","shell.execute_reply":"2022-08-28T09:56:43.067051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img_mask(train_df, func_for_id = max, col_for_func = \"img_height\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:43.073247Z","iopub.execute_input":"2022-08-28T09:56:43.074384Z","iopub.status.idle":"2022-08-28T09:56:46.947401Z","shell.execute_reply.started":"2022-08-28T09:56:43.074312Z","shell.execute_reply":"2022-08-28T09:56:46.946145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img_mask(train_df, func_for_id = max, col_for_func = \"img_width\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:46.948986Z","iopub.execute_input":"2022-08-28T09:56:46.949467Z","iopub.status.idle":"2022-08-28T09:56:50.231894Z","shell.execute_reply.started":"2022-08-28T09:56:46.949431Z","shell.execute_reply":"2022-08-28T09:56:50.230607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img_mask(train_df, func_for_id = min, col_for_func = \"img_width\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:50.233461Z","iopub.execute_input":"2022-08-28T09:56:50.233886Z","iopub.status.idle":"2022-08-28T09:56:52.203098Z","shell.execute_reply.started":"2022-08-28T09:56:50.233855Z","shell.execute_reply":"2022-08-28T09:56:52.201987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Case of 18121: what appears to be the mask intersecting with the border of the picture. After verifying that it is not due to a cropping error in diplaying the image, we should then check wether this is an odd case or a recurring commonplace occurence.","metadata":{}},{"cell_type":"code","source":"img = Image.open(f'{datadir}/train_images/18121.tiff')\nplt.figure(figsize=(6,6))\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:52.204538Z","iopub.execute_input":"2022-08-28T09:56:52.205089Z","iopub.status.idle":"2022-08-28T09:56:53.072907Z","shell.execute_reply.started":"2022-08-28T09:56:52.205055Z","shell.execute_reply":"2022-08-28T09:56:53.071583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is not a cropping error, let's verify  how many of these cases there are:","metadata":{}},{"cell_type":"code","source":"def is_mask_on_border(mask: np.array) -> bool:\n    top_arr = mask[0,:]\n    if 1 in top_arr:\n        return True\n    bottom_arr = mask[-1,:]\n    if 1 in bottom_arr:\n        return True\n    left_arr = mask[:,0]\n    if 1 in left_arr:\n        return True\n    right_arr = mask[:,-1]\n    if 1 in right_arr:\n        return True\n    return False","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:53.074934Z","iopub.execute_input":"2022-08-28T09:56:53.076044Z","iopub.status.idle":"2022-08-28T09:56:53.084002Z","shell.execute_reply.started":"2022-08-28T09:56:53.075995Z","shell.execute_reply":"2022-08-28T09:56:53.082786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_ids_to_check = []\nfor idx, line in train_df.iterrows():\n    mask = rle_decode(line['rle'], (line['img_height'], line['img_width']))\n    if is_mask_on_border(mask):\n        list_ids_to_check.append(line['id'])\nprint(f\"There are {len(list_ids_to_check)}/{len(train_df.index)} masks intersecting with image border\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:53.085951Z","iopub.execute_input":"2022-08-28T09:56:53.087201Z","iopub.status.idle":"2022-08-28T09:56:55.338484Z","shell.execute_reply.started":"2022-08-28T09:56:53.087132Z","shell.execute_reply":"2022-08-28T09:56:55.337102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def is_mask_next_to_border(mask: np.array) -> bool:\n    top_arr = mask[1,:]\n    if 1 in top_arr:\n        return True\n    bottom_arr = mask[-2,:]\n    if 1 in bottom_arr:\n        return True\n    left_arr = mask[:,1]\n    if 1 in left_arr:\n        return True\n    right_arr = mask[:,-2]\n    if 1 in right_arr:\n        return True\n    return False","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:55.340191Z","iopub.execute_input":"2022-08-28T09:56:55.340800Z","iopub.status.idle":"2022-08-28T09:56:55.349460Z","shell.execute_reply.started":"2022-08-28T09:56:55.340763Z","shell.execute_reply":"2022-08-28T09:56:55.348283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_ids_to_check = []\nfor idx, line in train_df.iterrows():\n    mask = rle_decode(line['rle'], (line['img_height'], line['img_width']))\n    if is_mask_next_to_border(mask):\n        list_ids_to_check.append(line['id'])\nprint(f\"There are {len(list_ids_to_check)}/{len(train_df.index)} masks next to image border\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:55.351037Z","iopub.execute_input":"2022-08-28T09:56:55.351435Z","iopub.status.idle":"2022-08-28T09:56:57.338649Z","shell.execute_reply.started":"2022-08-28T09:56:55.351402Z","shell.execute_reply":"2022-08-28T09:56:57.337057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_ids_to_check","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.340222Z","iopub.execute_input":"2022-08-28T09:56:57.340618Z","iopub.status.idle":"2022-08-28T09:56:57.348988Z","shell.execute_reply.started":"2022-08-28T09:56:57.340582Z","shell.execute_reply":"2022-08-28T09:56:57.347549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"for id_img in list_ids_to_check:\n     plot_img_mask(train_df, id_img = id_img)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.350584Z","iopub.execute_input":"2022-08-28T09:56:57.351017Z","iopub.status.idle":"2022-08-28T09:56:57.364540Z","shell.execute_reply.started":"2022-08-28T09:56:57.350981Z","shell.execute_reply":"2022-08-28T09:56:57.362621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelisation","metadata":{}},{"cell_type":"code","source":"train_df['img_path'] = train_df['id'].apply(lambda x: f\"{datadir}/train_images/{x}.tiff\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.365958Z","iopub.execute_input":"2022-08-28T09:56:57.366474Z","iopub.status.idle":"2022-08-28T09:56:57.377579Z","shell.execute_reply.started":"2022-08-28T09:56:57.366426Z","shell.execute_reply":"2022-08-28T09:56:57.375657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_tiff_tensor(img_path):\n    img = Image.open(img_path)\n    img = np.array(img).astype(np.float32) / 255\n    ti = tf.convert_to_tensor(img, dtype=tf.uint8)\n    return ti\n\ndef decode_tiff_experimental(img_path):\n    img = tf.io.read_file(img_path)\n    img = tfio.experimental.image.decode_tiff(img)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.378811Z","iopub.execute_input":"2022-08-28T09:56:57.379624Z","iopub.status.idle":"2022-08-28T09:56:57.390184Z","shell.execute_reply.started":"2022-08-28T09:56:57.379573Z","shell.execute_reply":"2022-08-28T09:56:57.389225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height, img_width = 1024, 1024","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.391757Z","iopub.execute_input":"2022-08-28T09:56:57.392932Z","iopub.status.idle":"2022-08-28T09:56:57.404474Z","shell.execute_reply.started":"2022-08-28T09:56:57.392887Z","shell.execute_reply":"2022-08-28T09:56:57.402214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode_tf(mask_rle, shape):\n    # Copied from https://stackoverflow.com/questions/58693261/decoding-rle-run-length-encoding-mask-with-tensorflow-datasets\n    shape = tf.convert_to_tensor(shape, tf.int64)\n    size = tf.math.reduce_prod(shape)\n    # Split string\n    s = tf.strings.split(mask_rle)\n    s = tf.strings.to_number(s, tf.int64)\n    # Get starts and lengths\n    starts = s[::2] - 1\n    lens = s[1::2]\n    # Make ones to be scattered\n    total_ones = tf.reduce_sum(lens)\n    ones = tf.ones([total_ones], tf.uint8)\n    # Make scattering indices\n    r = tf.range(total_ones)\n    lens_cum = tf.math.cumsum(lens)\n    s = tf.searchsorted(lens_cum, r, 'right')\n    idx = r + tf.gather(starts - tf.pad(lens_cum[:-1], [(1, 0)]), s)\n    # Scatter ones into flattened mask\n    mask_flat = tf.scatter_nd(tf.expand_dims(idx, 1), ones, [size])\n    # Change shape to add channel for resize (to delete if not working)\n    shape = (1, shape[0], shape[1])\n    # Reshape into mask\n    return tf.reshape(mask_flat, shape)\n\ndef resize(new_input):\n    new_input = tf.image.resize(new_input, (img_height, img_width), method=\"nearest\")\n    return new_input\n\ndef preprocessing(img_path: str, mask_rle: str, shape: tuple = (3000, 3000)): \n    img = decode_tiff_experimental(img_path) \n    img = tf.image.resize(img, shape)\n    img = tf.cast(img,  tf.float32) / 255.\n    img = img[:,:,:3] # Removing alpha channel\n    mask = tf.transpose(rle_decode_tf(mask_rle, shape))\n    \n    img = resize(img)    \n    mask = resize(mask)\n    return img, mask\n\ndef get_data(df: pd.DataFrame) -> Dataset:\n    train_df, test_df = train_test_split(df, stratify=df['organ'], train_size=0.7)\n    ds_train = Dataset.from_tensor_slices((train_df['img_path'].values, train_df['rle'].values))\n    ds_train = ds_train.map(preprocessing, tf.data.experimental.AUTOTUNE)\n    ds_test = Dataset.from_tensor_slices((test_df['img_path'].values, test_df['rle'].values))\n    ds_test = ds_test.map(preprocessing, tf.data.experimental.AUTOTUNE)\n    return ds_train, ds_test\n\n########################################################################","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.406649Z","iopub.execute_input":"2022-08-28T09:56:57.406990Z","iopub.status.idle":"2022-08-28T09:56:57.425398Z","shell.execute_reply.started":"2022-08-28T09:56:57.406961Z","shell.execute_reply":"2022-08-28T09:56:57.424186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_values_function(img_total_height: int, img_total_width: int, output_size: int, with_overlap: bool=False) -> list:\n    \"\"\"\n    with given height and width, return list of offset values for tf.image.crop_to_bounding_box\n    by default, no overlap, meaning that two adjacent borders of width: total % output_size will be cropped out.\n    with overlap, boundaries are moved of only half the value (approximately doubling the length of returned list).\n    \"\"\"\n    # Creating lists:\n    list_height_offset = [x for x in range(0, img_total_height - output_size, output_size + 1)]\n    list_width_offset = [x for x in range(0, img_total_width - output_size, output_size + 1)]\n    if not with_overlap:\n        # itertool product gives us the result\n        list_offset = list(product(list_height_offset, list_width_offset))\n        return list_offset\n    # Adding overlapping of images, as well as border and corner case\n    if not output_size % 2 == 0:\n        raise ValueError(f\"output_size must be even, not odd: {output_size}\")\n    list_height_offset.extend([x for x in range(int(output_size/2), img_total_height - output_size, output_size + 1)])\n    list_width_offset.extend([x for x in range(int(output_size/2), img_total_width - output_size, output_size + 1)])\n    # Sorting values:\n    list_height_offset.sort()\n    list_width_offset.sort()\n    # Appending last value:\n    list_height_offset.append(img_total_height - output_size - 1)\n    list_width_offset.append(img_total_width - output_size - 1)\n    # Result\n    list_offset = list(product(list_height_offset, list_width_offset))\n    return list_offset\n\ndef get_df_for_crop(df: pd.DataFrame, output_size: int = 256, with_overlap: bool=False) -> pd.DataFrame:\n    \"\"\"\n    Returns dataframe with columns offset_height and offset_width for cropping\n    \"\"\"\n    new_df = df.copy()\n    new_df['tuple_temp'] = new_df.apply(lambda x: crop_values_function(x['img_height'], \n                                                                       x['img_width'],\n                                                                       output_size=output_size, \n                                                                       with_overlap=with_overlap),\n                                               axis=1)\n    new_df = new_df.explode('tuple_temp')\n    new_df['offset_height'] = new_df['tuple_temp'].apply(lambda x: x[0])\n    new_df['offset_width'] = new_df['tuple_temp'].apply(lambda x: x[1])\n    new_df.drop('tuple_temp', axis=1, inplace=True)\n    return new_df","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.426901Z","iopub.execute_input":"2022-08-28T09:56:57.427315Z","iopub.status.idle":"2022-08-28T09:56:57.445724Z","shell.execute_reply.started":"2022-08-28T09:56:57.427274Z","shell.execute_reply":"2022-08-28T09:56:57.444567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing_crop(img_path: str, mask_rle: str, target_height:int = 3000, target_width:int = 3000): \n    \"\"\"\n    Huge thanks to hoyso48 for the help!  (https://www.kaggle.com/competitions/hubmap-organ-segmentation/discussion/347568)\n    LorSong as well (https://github.com/tensorflow/tensorflow/issues/6743)\n    \n    \"\"\"\n    # Decode entire image\n    img = decode_tiff_experimental(img_path) \n    # increase contrast\n    #img = tf.image.adjust_contrast(img, 2.)\n    # Set data between 0 and 1\n    img = tf.cast(img,  tf.float32) / 255.\n    # removing alpha channel\n    img = img[:,:,:3]\n    # resizing\n    img = resize(img)\n    # grayscaling, \n    img = tf.image.rgb_to_grayscale(img)\n\n    # Decode entire mask\n    mask = tf.transpose(rle_decode_tf(mask_rle, (target_width, target_height)))\n    # resize mask\n    mask = resize(mask)\n\n    # cropping image:\n    img = tf.expand_dims(img, axis=0)\n    img = tf.image.extract_patches(images=img, \n                                  sizes=[1, 256, 256, 1],\n                                  strides=[1, 256, 256 ,1],\n                                  rates=[1, 1, 1, 1],\n                                  padding='SAME')\n    # reshaping into image\n    img = tf.reshape(img, (-1, 256,256))\n    # cropping mask:\n    mask = tf.expand_dims(mask, axis=0)\n    mask = tf.image.extract_patches(images=mask, \n                                  sizes=[1, 256, 256, 1],\n                                  strides=[1, 256, 256 ,1],\n                                  rates=[1, 1, 1, 1],\n                                  padding='VALID')\n    # reshaping into image\n    mask = tf.reshape(mask, (-1, 256,256))\n    return img, mask\n","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.447458Z","iopub.execute_input":"2022-08-28T09:56:57.447876Z","iopub.status.idle":"2022-08-28T09:56:57.468970Z","shell.execute_reply.started":"2022-08-28T09:56:57.447828Z","shell.execute_reply":"2022-08-28T09:56:57.466240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data_cropped(df: pd.DataFrame) -> Dataset:\n    \"\"\"\n    \"\"\"\n    # Split data between test and train sets\n    train_df, test_df = train_test_split(df, train_size=0.7, stratify=df['organ'])\n    # deleting unused columns to free space\n    train_df.drop(['data_source', 'pixel_size', 'tissue_thickness', 'age','sex'], axis=1, inplace=True)\n    test_df.drop(['data_source', 'pixel_size', 'tissue_thickness', 'age','sex'], axis=1, inplace=True)\n    # Create dataset, then apply preprocessing function\n    ds_train = Dataset.from_tensor_slices((train_df['img_path'].values, \n                                           train_df['rle'].values, \n                                           train_df['img_height'].values,\n                                           train_df['img_width'].values))\n    ds_train = ds_train.map(preprocessing_crop, tf.data.experimental.AUTOTUNE)\n    # same for testing\n    ds_test = Dataset.from_tensor_slices((test_df['img_path'].values, \n                                          test_df['rle'].values, \n                                          test_df['img_height'].values, \n                                          test_df['img_width'].values))\n    ds_test = ds_test.map(preprocessing_crop, tf.data.experimental.AUTOTUNE)\n    return ds_train, ds_test, test_df","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.470901Z","iopub.execute_input":"2022-08-28T09:56:57.472165Z","iopub.status.idle":"2022-08-28T09:56:57.488733Z","shell.execute_reply.started":"2022-08-28T09:56:57.472095Z","shell.execute_reply":"2022-08-28T09:56:57.487358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train, ds_test, test_df = get_data_cropped(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:57.490840Z","iopub.execute_input":"2022-08-28T09:56:57.492463Z","iopub.status.idle":"2022-08-28T09:56:58.718383Z","shell.execute_reply.started":"2022-08-28T09:56:57.492400Z","shell.execute_reply":"2022-08-28T09:56:58.717052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:58.724684Z","iopub.execute_input":"2022-08-28T09:56:58.725883Z","iopub.status.idle":"2022-08-28T09:56:58.731272Z","shell.execute_reply.started":"2022-08-28T09:56:58.725840Z","shell.execute_reply":"2022-08-28T09:56:58.729995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From https://pyimagesearch.com/2022/02/21/u-net-image-segmentation-in-keras/\ndef double_conv_block(x, n_filters):\n    # Conv2D then ReLU activation\n    #x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n    x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n    # Conv2D then ReLU activation\n    #x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n    x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n\n    return x\n\ndef downsample_block(x, n_filters):\n    f = double_conv_block(x, n_filters)\n    p = layers.MaxPool2D(2)(f)\n    p = layers.Dropout(0.3)(p)\n    return f, p\n\ndef upsample_block(x, conv_features, n_filters):\n    # upsample\n    #x = layers.Conv2DTranspose(n_filters, 3, 2, padding=\"same\")(x)\n    x = layers.Conv2DTranspose(n_filters, 1, 2, padding=\"same\")(x)\n    # concatenate\n    x = layers.concatenate([x, conv_features])\n    # dropout\n    x = layers.Dropout(0.3)(x)\n    # Conv2D twice with ReLU activation\n    x = double_conv_block(x, n_filters)\n    return x\n\ndef get_unet_model():\n    # inputs\n    #inputs = layers.Input(shape=(128,128,3))\n    inputs = layers.Input(shape=(256,256,1))\n    # encoder: contracting path - downsample\n    # 1 - downsample\n    f1, p1 = downsample_block(inputs, 64)\n    # 2 - downsample\n    f2, p2 = downsample_block(p1, 128)\n    # 3 - downsample\n    f3, p3 = downsample_block(p2, 256)\n    # 4 - downsample\n    f4, p4 = downsample_block(p3, 512)\n    # 5 - bottleneck\n    bottleneck = double_conv_block(p4, 1024)\n    # decoder: expanding path - upsample\n    # 6 - upsample\n    u6 = upsample_block(bottleneck, f4, 512)\n    # 7 - upsample\n    u7 = upsample_block(u6, f3, 256)\n    # 8 - upsample\n    u8 = upsample_block(u7, f2, 128)\n    # 9 - upsample\n    u9 = upsample_block(u8, f1, 64)\n    # outputs\n    #outputs = layers.Conv2D(3, 1, padding=\"same\", activation = \"softmax\")(u9)\n    outputs = layers.Conv2D(1, 1, padding=\"same\", activation = \"sigmoid\")(u9)\n\n    # unet model with Keras Functional API\n    unet_model = tf.keras.Model(inputs, outputs, name=\"U-Net\")\n    return unet_model","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:58.733299Z","iopub.execute_input":"2022-08-28T09:56:58.733660Z","iopub.status.idle":"2022-08-28T09:56:58.749477Z","shell.execute_reply.started":"2022-08-28T09:56:58.733629Z","shell.execute_reply":"2022-08-28T09:56:58.748396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unet_model = get_unet_model()\nunet_model","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:58.751321Z","iopub.execute_input":"2022-08-28T09:56:58.752490Z","iopub.status.idle":"2022-08-28T09:56:59.783322Z","shell.execute_reply.started":"2022-08-28T09:56:58.752438Z","shell.execute_reply":"2022-08-28T09:56:59.782396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unet_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:59.784706Z","iopub.execute_input":"2022-08-28T09:56:59.785589Z","iopub.status.idle":"2022-08-28T09:56:59.797606Z","shell.execute_reply.started":"2022-08-28T09:56:59.785552Z","shell.execute_reply":"2022-08-28T09:56:59.795796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unet_model.compile(optimizer=tf.keras.optimizers.SGD(learning_rate=0.009, momentum=0.9),\n                    loss=tf.keras.losses.BinaryCrossentropy(),\n                    metrics=tf.keras.metrics.MeanIoU(num_classes=2)\n                  )","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:59.799429Z","iopub.execute_input":"2022-08-28T09:56:59.800615Z","iopub.status.idle":"2022-08-28T09:56:59.827140Z","shell.execute_reply.started":"2022-08-28T09:56:59.800573Z","shell.execute_reply":"2022-08-28T09:56:59.826260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 32\nBUFFER_SIZE = 256\n\"\"\"train_batches = ds_train.shuffle(BUFFER_SIZE).batch(BATCH_SIZE).cache().repeat()\ntrain_batches = train_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\ntest_batches = ds_test.shuffle(BUFFER_SIZE).batch(BATCH_SIZE).cache().repeat()\ntest_batches = test_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\"\"\"\n\"\"\"train_batches = ds_train.batch(BATCH_SIZE).shuffle(BUFFER_SIZE).cache().repeat()\ntrain_batches = train_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\ntest_batches = ds_test.batch(BATCH_SIZE).shuffle(BUFFER_SIZE).cache().repeat()\ntest_batches = test_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\"\"\"\n\ntrain_batches = ds_train.unbatch().shuffle(BUFFER_SIZE).batch(BATCH_SIZE).cache().repeat()\ntrain_batches = train_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\ntest_batches = ds_test.unbatch().shuffle(BUFFER_SIZE).batch(BATCH_SIZE).cache().repeat()\ntest_batches = test_batches.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:59.828634Z","iopub.execute_input":"2022-08-28T09:56:59.830009Z","iopub.status.idle":"2022-08-28T09:56:59.865575Z","shell.execute_reply.started":"2022-08-28T09:56:59.829972Z","shell.execute_reply":"2022-08-28T09:56:59.864418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = unet_model.fit(train_batches,\n                                validation_data = test_batches,\n                                epochs=15,\n                                steps_per_epoch=5,  \n                                validation_steps=5\n                                )","metadata":{"execution":{"iopub.status.busy":"2022-08-28T09:56:59.867086Z","iopub.execute_input":"2022-08-28T09:56:59.867578Z","iopub.status.idle":"2022-08-28T10:19:07.723322Z","shell.execute_reply.started":"2022-08-28T09:56:59.867532Z","shell.execute_reply":"2022-08-28T10:19:07.720857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_batches\ndel test_batches\n#TEMPORARY! trying to save ram space...\ndel model_history","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:07.727107Z","iopub.execute_input":"2022-08-28T10:19:07.727676Z","iopub.status.idle":"2022-08-28T10:19:07.735151Z","shell.execute_reply.started":"2022-08-28T10:19:07.727591Z","shell.execute_reply":"2022-08-28T10:19:07.734201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:07.736363Z","iopub.execute_input":"2022-08-28T10:19:07.737035Z","iopub.status.idle":"2022-08-28T10:19:08.120322Z","shell.execute_reply.started":"2022-08-28T10:19:07.737004Z","shell.execute_reply":"2022-08-28T10:19:08.119003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unet_model.save('./unet_model')","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.122076Z","iopub.execute_input":"2022-08-28T10:19:08.122943Z","iopub.status.idle":"2022-08-28T10:19:08.131246Z","shell.execute_reply.started":"2022-08-28T10:19:08.122890Z","shell.execute_reply":"2022-08-28T10:19:08.130318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"import keras\nunet_model = keras.models.load_model('./unet_model')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.132494Z","iopub.execute_input":"2022-08-28T10:19:08.133488Z","iopub.status.idle":"2022-08-28T10:19:08.148073Z","shell.execute_reply.started":"2022-08-28T10:19:08.133451Z","shell.execute_reply":"2022-08-28T10:19:08.146805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"prediction = unet_model.predict(np.reshape(img, (1, 128, 128, 3)))\nprediction.shape\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.149716Z","iopub.execute_input":"2022-08-28T10:19:08.150901Z","iopub.status.idle":"2022-08-28T10:19:08.163437Z","shell.execute_reply.started":"2022-08-28T10:19:08.150860Z","shell.execute_reply":"2022-08-28T10:19:08.162161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.python.ops import math_ops, array_ops\n\ndef simple_rgb_to_grayscale(image: np.array):\n    # Same weights used by tensorflow for tf.image.rgb_to_grayscale\n    rgb_weights = [0.2989, 0.5870, 0.1140]\n    gray_image = image.numpy().dot(rgb_weights)\n    return tf.convert_to_tensor(gray_image)\n\ndef get_values_for_crop(img_total_height: int = 1024, img_total_width: int = 1024, output_size: int = 256) -> dict:\n    \"\"\"\n    with given height and width, return dictionary of offset values. keys are heights, values are lists of widths.\n    \"\"\"\n    # Creating lists:\n    list_height_offset = [x for x in range(0, img_total_height - output_size + 1, output_size)]\n    list_width_offset = [x for x in range(0, img_total_width - output_size + 1, output_size)]\n    dict_hw = dict(zip(list_height_offset, [[] for x in range(len(list_height_offset))]))\n    for h in list_height_offset:\n        for w in list_width_offset:\n            dict_hw[h].append(w)\n    return dict_hw\n\ndef get_data_for_submit() -> pd.DataFrame:\n    datadir = \"/kaggle/input/hubmap-organ-segmentation/\"\n    df_for_predict = pd.read_csv(datadir+\"test.csv\")\n    df_for_predict['img_path'] = df_for_predict['id'].apply(lambda x: f\"{datadir}/test_images/{x}.tiff\")\n    return df_for_predict\n\ndef get_submit_csv(data_for_pred: pd.DataFrame, model):\n    datadir = \"/kaggle/input/hubmap-organ-segmentation\"\n    shape_val = 256\n    result_df = pd.DataFrame(columns=['id', 'rle'])\n    for idx, line in data_for_pred.iterrows():\n        img = decode_tiff_experimental(f\"{datadir}/test_images/{line['id']}.tiff\") \n        # removing alpha channel\n        img = img[:,:,:3]\n        # resizing\n        img = resize(img)\n        # grayscaling \n        img = simple_rgb_to_grayscale(img)\n        img = tf.cast(img,  tf.float32) / 255.\n        mask_pred = None\n        val_temp = None\n        for height, list_width in get_values_for_crop().items():\n            array_temp = None\n            for width in list_width:\n                img_crop_for_pred = img[height:height+shape_val,\n                                                        width:width+shape_val]\n                pred_from_crop =  model.predict(np.expand_dims(img_crop_for_pred, axis=0))\n                if array_temp is None:\n                    array_temp = pred_from_crop[0]\n                else:\n                    array_temp = np.hstack([array_temp, pred_from_crop[0]])\n            if mask_pred is None:\n                mask_pred = array_temp\n            else:\n                mask_pred = np.vstack([mask_pred, array_temp])\n        # pred to mask\n        val = 0.18\n        mask_pred[mask_pred < val] = 0\n        mask_pred[mask_pred > 0] = 1\n        result_df = result_df.append(pd.Series([line['id'], rle_encode(mask_pred)], index=['id', 'rle']), ignore_index=True)\n    result_df.to_csv('./submission.csv', index=False)\n    return result_df","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:30:59.860498Z","iopub.execute_input":"2022-08-28T10:30:59.860977Z","iopub.status.idle":"2022-08-28T10:30:59.879831Z","shell.execute_reply.started":"2022-08-28T10:30:59.860941Z","shell.execute_reply":"2022-08-28T10:30:59.878321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_submit_csv(get_data_for_submit(), unet_model)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:31:09.626053Z","iopub.execute_input":"2022-08-28T10:31:09.626618Z","iopub.status.idle":"2022-08-28T10:31:29.675761Z","shell.execute_reply.started":"2022-08-28T10:31:09.626554Z","shell.execute_reply":"2022-08-28T10:31:29.674148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"shape_val = 256\ntest_df = get_df_for_crop(test_df)\n\nfor n, (id_img, frame) in enumerate(test_df.groupby('id')):\n    if n > 5:\n        break\n    img = Image.open(f\"{datadir}/train_images/{id_img}.tiff\")\n    img = np.array(img).astype(np.float32) / 255.\n    img = tf.image.rgb_to_grayscale(img)\n    img = tf.image.resize(img, (1024, 1024), method=\"nearest\")\n    mask_pred = None\n    val_temp = None\n    for offset_height, frame2 in frame.groupby('offset_height'):\n        array_temp = None\n        for idx, line in frame2.iterrows():\n            img_crop_for_pred = img[line['offset_width']:line['offset_width']+shape_val,\n                                                  line['offset_height']:line['offset_height']+shape_val,\n                                                  :]\n            pred_from_crop =  tf.transpose(unet_model.predict(np.expand_dims(img_crop_for_pred, axis=0)))\n            if array_temp is None:\n                array_temp = pred_from_crop[0]\n            else:\n                array_temp = np.append(array_temp, pred_from_crop[0], axis = 1)\n        if mask_pred is None:\n            mask_pred = array_temp\n        else:\n            mask_pred = np.append(mask_pred, array_temp, axis=0)\n    fig, axs = plt.subplots(ncols=3, figsize=(20, 20))\n    mask_true = rle_decode(line['rle'], (line['img_height'], line['img_width'])).astype(np.float32)\n    axs[0].imshow(mask_true)\n    axs[1].imshow(img)\n    '''val = 0.05\n    mask_pred[mask_pred < val_pred] = 0\n    mask_pred[mask_pred >= val_pred] = 1'''\n    axs[2].imshow(mask_pred.T[0])\n    plt.show()\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.343446Z","iopub.status.idle":"2022-08-28T10:19:08.344051Z","shell.execute_reply.started":"2022-08-28T10:19:08.343750Z","shell.execute_reply":"2022-08-28T10:19:08.343779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"def plot_comparison(model, ds_test, n: int=0, val_pred: float=0.5) -> None:\n    fig, axs = plt.subplots(ncols=3, figsize=(20, 20))\n    # Access Dataset values:\n    for i, (img, mask) in enumerate(ds_test):\n        if i == n:\n            plot_img = img\n            plot_mask = mask\n            break\n    axs[0].imshow(plot_mask)\n    axs[1].imshow(plot_img)\n    prediction = model.predict(np.reshape(img, (1, 128, 128, 3)))\n    val = 0.8\n    prediction[prediction < val_pred] = 0\n    prediction[prediction >= val_pred] = 1\n    axs[2].imshow(prediction[0])\n    plt.show()\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.345647Z","iopub.status.idle":"2022-08-28T10:19:08.346083Z","shell.execute_reply.started":"2022-08-28T10:19:08.345886Z","shell.execute_reply":"2022-08-28T10:19:08.345905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"for n in range(3):\n    for val_pred in [0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7]:\n        plot_comparison(unet_model, ds_test, n = n, val_pred=val_pred)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:19:08.347780Z","iopub.status.idle":"2022-08-28T10:19:08.348306Z","shell.execute_reply.started":"2022-08-28T10:19:08.348091Z","shell.execute_reply":"2022-08-28T10:19:08.348113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}