{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This Kaggle Dataset has 351 images of 3000x3000px images. As this image size can't be executed in most of the GPUs, we will explore the solution of creating a dataset dividing each image in tiles.\nMoreover, we explore all the data available in this dataset, and create a PyTorch Dataset Class to work with this competition.\n\nThis work is heavily inspired by my fellow teammate Miquel: https://www.kaggle.com/code/miquel0/hubmap-hba-tiled-dataset\n\n# Table of contents\n\n* [Imports](#section-one)\n* [Config](#section-one0)\n* [Dataset Data](#section-one1)\n* [Train Dataset CSV] (#section-one12)\n* [Utils](#section-two)\n* [Tiled dataset](#section-three)\n* [Plot example](#section-three)\n","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom PIL import Image\nimport glob\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom skimage import color\nimport cv2\nfrom tqdm import tqdm\nimport os\nimport wandb\nfrom kaggle_secrets import UserSecretsClient\nimport seaborn as sns\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:17:01.990212Z","iopub.execute_input":"2022-07-27T10:17:01.990975Z","iopub.status.idle":"2022-07-27T10:17:02.544299Z","shell.execute_reply.started":"2022-07-27T10:17:01.990891Z","shell.execute_reply":"2022-07-27T10:17:02.541694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one0\"></a>\n# Config","metadata":{}},{"cell_type":"code","source":"class config:\n    BASE_PATH = \"../input/hubmap-organ-segmentation/\"\n    TRAIN_PATH = os.path.join(BASE_PATH, \"train\")\n\nuser_secrets = UserSecretsClient()\napi_key = user_secrets.get_secret(\"WANDB\")\nwandb.login(key=api_key)\nwandb.init()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T10:04:15.575501Z","iopub.execute_input":"2022-07-27T10:04:15.576057Z","iopub.status.idle":"2022-07-27T10:04:20.356354Z","shell.execute_reply.started":"2022-07-27T10:04:15.576028Z","shell.execute_reply":"2022-07-27T10:04:20.355536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# Utils\n\nBoth rle encoding and decoding functions are from this kaggle notebook: https://www.kaggle.com/code/ishandutta/hubmap-complete-understanding-and-eda-w-b","metadata":{}},{"cell_type":"code","source":"def mask2rle(img):\n    \"\"\"\n    :param img: numpy array, 1 - mask, 0 - background\n    :return: run length as string formated\n    \"\"\"\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef rle2mask(mask_rle, shape=(1600, 256)):\n    \"\"\"\n    :param mask_rle: run-length as string formated (start length)\n    :param shape: (width,height) of array to return\n    :return: numpy array, 1 - mask, 0 - background\n    \"\"\"\n\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:05:51.821065Z","iopub.execute_input":"2022-07-27T10:05:51.821471Z","iopub.status.idle":"2022-07-27T10:05:51.834703Z","shell.execute_reply.started":"2022-07-27T10:05:51.821442Z","shell.execute_reply":"2022-07-27T10:05:51.833040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one1\"></a>\n# Dataset Data","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-one12\"></a>\n## Train Dataset CSV","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\n    os.path.join(config.BASE_PATH, \"train.csv\")\n)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:01:58.824822Z","iopub.execute_input":"2022-07-27T10:01:58.825236Z","iopub.status.idle":"2022-07-27T10:01:59.285680Z","shell.execute_reply.started":"2022-07-27T10:01:58.825205Z","shell.execute_reply":"2022-07-27T10:01:59.284567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the train dataset is from HPA. Take into account that the test dataset will be from another source of images!\n\nAll images are 3000x3000px images (from HPA). The HuBMAP images range in size from 4500x4500 down to 160x160 pixels. So its important to tile our dataset, and generalize when performing mask detection.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\ng = sns.countplot(data=df, x=\"organ\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"Organ Counts\", color = \"black\")\n\nplt.figure(figsize=(15, 5))\ng = sns.histplot(data=df, x=\"age\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"Age\", color = \"black\")\n\nplt.figure(figsize=(15, 5))\ng = sns.histplot(data=df, x=\"sex\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"Sex\", color = \"black\")\n\nplt.figure(figsize=(15, 5))\ng = sns.histplot(data=df, x=\"data_source\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"data_source\", color = \"black\")\n\nplt.figure(figsize=(15, 5))\ng = sns.histplot(data=df, x=\"pixel_size\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"pixel_size\", color = \"black\")\n\nplt.figure(figsize=(15, 5))\ng = sns.histplot(data=df, x=\"tissue_thickness\", palette=sns.color_palette(\"Set2\", 8))\ng.set_title(\"tissue_thickness\", color = \"black\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:22:27.329405Z","iopub.execute_input":"2022-07-27T10:22:27.330076Z","iopub.status.idle":"2022-07-27T10:22:28.783863Z","shell.execute_reply.started":"2022-07-27T10:22:27.330044Z","shell.execute_reply":"2022-07-27T10:22:28.780808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one13\"></a>\n## Example Image and Mask","metadata":{}},{"cell_type":"code","source":"img_id_1 = 10274\nimg_1 = cv2.imread(config.BASE_PATH + \"train_images/\" + str(img_id_1) + \".tiff\")\nprint('Input Shape:', img_1.shape)\nmask_1 = rle2mask(df[df[\"id\"]==img_id_1][\"rle\"].iloc[-1], (img_1.shape[1], img_1.shape[0]))\nprint('Mask Shape:', mask_1.shape)\n\n\nfig, (ax1, ax2) = plt.subplots(1, 2,figsize = (15,15))\n#fig.suptitle('Train images')\nax1.imshow(img_1)\nax2.imshow(mask_1, cmap = \"PRGn\", alpha=0.5)\nplt.axis(\"off\")\nfig.tight_layout()\nfig.subplots_adjust(top=0.95)\n\nwandb.log({\"Image Sample 1\": fig})","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:14:45.349664Z","iopub.execute_input":"2022-07-27T10:14:45.350035Z","iopub.status.idle":"2022-07-27T10:14:49.190531Z","shell.execute_reply.started":"2022-07-27T10:14:45.350007Z","shell.execute_reply":"2022-07-27T10:14:49.188360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nplt.imshow(img_1)\nplt.imshow(mask_1, cmap='PRGn', alpha=0.5)\nplt.axis(\"off\")\nwandb.log({\"Image with Mask\": plt})","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:14:37.368712Z","iopub.execute_input":"2022-07-27T10:14:37.369151Z","iopub.status.idle":"2022-07-27T10:14:42.228092Z","shell.execute_reply.started":"2022-07-27T10:14:37.369120Z","shell.execute_reply":"2022-07-27T10:14:42.226893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}