{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Goals**\n\nThe goal of this competition is to identify the locations of each functional tissue unit (FTU) in biopsy slides from several different organs (prostate, lungs, kidney, spleen...). The underlying data includes imagery from different sources prepared with different protocols at a variety of resolutions, reflecting typical challenges for working with medical data.\n\n\n1. [Exploratory Data Analysis](#Exploratory-Data-Analysis)\n2. [Image and Mask Visualization](#Image-and-Mask-Visualization)\n3. [Modelling & Inference](#Modelling-&-Inference)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport cv2\nimport os, random\nimport tifffile as tiff \nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-03T20:56:57.672835Z","iopub.execute_input":"2022-09-03T20:56:57.673293Z","iopub.status.idle":"2022-09-03T20:56:57.684057Z","shell.execute_reply.started":"2022-09-03T20:56:57.673253Z","shell.execute_reply":"2022-09-03T20:56:57.683040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"**Train Data**","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:38.369452Z","iopub.execute_input":"2022-09-03T19:28:38.370221Z","iopub.status.idle":"2022-09-03T19:28:38.946980Z","shell.execute_reply.started":"2022-09-03T19:28:38.370146Z","shell.execute_reply":"2022-09-03T19:28:38.937995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:38.948043Z","iopub.execute_input":"2022-09-03T19:28:38.948630Z","iopub.status.idle":"2022-09-03T19:28:38.996463Z","shell.execute_reply.started":"2022-09-03T19:28:38.948590Z","shell.execute_reply":"2022-09-03T19:28:38.980527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:38.999124Z","iopub.execute_input":"2022-09-03T19:28:38.999451Z","iopub.status.idle":"2022-09-03T19:28:39.030581Z","shell.execute_reply.started":"2022-09-03T19:28:38.999420Z","shell.execute_reply":"2022-09-03T19:28:39.021553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['organ'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:39.032146Z","iopub.execute_input":"2022-09-03T19:28:39.032672Z","iopub.status.idle":"2022-09-03T19:28:39.048297Z","shell.execute_reply.started":"2022-09-03T19:28:39.032637Z","shell.execute_reply":"2022-09-03T19:28:39.047009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 6))\nsns.countplot(data=df, x=\"organ\",alpha = 0.4).set_title(\"Organ Counts\")","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:39.049339Z","iopub.execute_input":"2022-09-03T19:28:39.049643Z","iopub.status.idle":"2022-09-03T19:28:39.454367Z","shell.execute_reply.started":"2022-09-03T19:28:39.049614Z","shell.execute_reply":"2022-09-03T19:28:39.453097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:39.455969Z","iopub.execute_input":"2022-09-03T19:28:39.456665Z","iopub.status.idle":"2022-09-03T19:28:39.467556Z","shell.execute_reply.started":"2022-09-03T19:28:39.456621Z","shell.execute_reply":"2022-09-03T19:28:39.465969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 6))\nsns.histplot(data=df, x=\"age\",alpha = 0.3).set_title(\"Age\")","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:39.469921Z","iopub.execute_input":"2022-09-03T19:28:39.470291Z","iopub.status.idle":"2022-09-03T19:28:39.798479Z","shell.execute_reply.started":"2022-09-03T19:28:39.470257Z","shell.execute_reply":"2022-09-03T19:28:39.797453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 6))\nsns.countplot(data=df, x=\"age\",hue = \"sex\",alpha = 0.3).set_title(\"Age\")","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:39.799653Z","iopub.execute_input":"2022-09-03T19:28:39.800439Z","iopub.status.idle":"2022-09-03T19:28:40.334668Z","shell.execute_reply.started":"2022-09-03T19:28:39.800401Z","shell.execute_reply":"2022-09-03T19:28:40.333705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 6))\nsns.histplot(data=df, x=\"sex\",alpha = 0.3).set_title(\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:40.338713Z","iopub.execute_input":"2022-09-03T19:28:40.339713Z","iopub.status.idle":"2022-09-03T19:28:40.615119Z","shell.execute_reply.started":"2022-09-03T19:28:40.339675Z","shell.execute_reply":"2022-09-03T19:28:40.613650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test Data**","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/hubmap-organ-segmentation/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:40.616668Z","iopub.execute_input":"2022-09-03T19:28:40.617521Z","iopub.status.idle":"2022-09-03T19:28:40.638288Z","shell.execute_reply.started":"2022-09-03T19:28:40.617483Z","shell.execute_reply":"2022-09-03T19:28:40.637141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**sample_submission**","metadata":{}},{"cell_type":"code","source":"submision_sample = pd.read_csv('../input/hubmap-organ-segmentation/sample_submission.csv')\nsubmision_sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:40.639958Z","iopub.execute_input":"2022-09-03T19:28:40.640308Z","iopub.status.idle":"2022-09-03T19:28:40.655230Z","shell.execute_reply.started":"2022-09-03T19:28:40.640272Z","shell.execute_reply":"2022-09-03T19:28:40.654214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image and Mask Visualization","metadata":{}},{"cell_type":"markdown","source":"**Train Images**","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/paulorzp/rle-functions-run-length-encode-decode\ndef mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n \ndef rle2mask(mask_rle, shape=(1600,256)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T\n","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:40.658089Z","iopub.execute_input":"2022-09-03T19:28:40.660601Z","iopub.status.idle":"2022-09-03T19:28:40.670726Z","shell.execute_reply.started":"2022-09-03T19:28:40.660565Z","shell.execute_reply":"2022-09-03T19:28:40.668992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organs = df['organ'].unique()\nfor organ in organs:\n    df_organ = df.loc[df['organ'] == organ].reset_index(drop=True)\n    plt.figure(figsize=(16,4))\n    i = 0\n    while i < 4:\n        img = tiff.imread( \"../input/hubmap-organ-segmentation/train_images/\" + str(df_organ['id'][i]) +'.tiff')\n        mask = rle2mask(df_organ['rle'][i], (img.shape[1], img.shape[0]))\n        plt.subplot(1, 4, i+1)\n        plt.imshow(img)\n        plt.imshow(mask, cmap='seismic', alpha=0.5)\n        plt.axis(\"off\")\n        i += 1\n    plt.suptitle(organ, fontsize=20)\n    plt.tight_layout()             ","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:28:40.673674Z","iopub.execute_input":"2022-09-03T19:28:40.674056Z","iopub.status.idle":"2022-09-03T19:29:29.080963Z","shell.execute_reply.started":"2022-09-03T19:28:40.673994Z","shell.execute_reply":"2022-09-03T19:29:29.079909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Target Image**","metadata":{}},{"cell_type":"code","source":"img = cv2.imread('../input/hubmap-organ-segmentation/test_images/10078.tiff')\nplt.figure(figsize=(9,9))\nplt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB)); plt.axis(\"off\");","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:29.082639Z","iopub.execute_input":"2022-09-03T19:29:29.083014Z","iopub.status.idle":"2022-09-03T19:29:29.973659Z","shell.execute_reply.started":"2022-09-03T19:29:29.082977Z","shell.execute_reply":"2022-09-03T19:29:29.971240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling & Inference","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastai.callback.all import *\nfrom fastai.basics import *\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:29.975325Z","iopub.execute_input":"2022-09-03T19:29:29.976457Z","iopub.status.idle":"2022-09-03T19:29:32.235511Z","shell.execute_reply.started":"2022-09-03T19:29:29.976401Z","shell.execute_reply":"2022-09-03T19:29:32.234368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = Path(\"../input/hubmap-2022-256x256/train\")\nlabel = Path(\"../input/hubmap-2022-256x256/masks\")\nlen(train.ls()), len(label.ls())","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:32.237022Z","iopub.execute_input":"2022-09-03T19:29:32.237844Z","iopub.status.idle":"2022-09-03T19:29:32.868634Z","shell.execute_reply.started":"2022-09-03T19:29:32.237802Z","shell.execute_reply":"2022-09-03T19:29:32.867741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_y(x): return label/x.name","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:32.871135Z","iopub.execute_input":"2022-09-03T19:29:32.871751Z","iopub.status.idle":"2022-09-03T19:29:32.877300Z","shell.execute_reply.started":"2022-09-03T19:29:32.871712Z","shell.execute_reply":"2022-09-03T19:29:32.876208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = SegmentationDataLoaders.from_label_func (train, \n                                         fnames = get_image_files(train),\n                                         label_func = get_y,\n                                         valid_pct=0.2, seed=None,\n                                         codes=['Bkgd', 'Cell'], \n                                         item_tfms=None,\n                                         batch_tfms=None, \n                                         bs = 6,\n                                        )","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:32.880447Z","iopub.execute_input":"2022-09-03T19:29:32.880825Z","iopub.status.idle":"2022-09-03T19:29:37.097325Z","shell.execute_reply.started":"2022-09-03T19:29:32.880798Z","shell.execute_reply":"2022-09-03T19:29:37.096195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"dls.show_batch(max_n=20)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:37.098886Z","iopub.execute_input":"2022-09-03T19:29:37.099545Z","iopub.status.idle":"2022-09-03T19:29:37.605257Z","shell.execute_reply.started":"2022-09-03T19:29:37.099507Z","shell.execute_reply":"2022-09-03T19:29:37.604382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = unet_learner(dls, resnet34 , metrics=Dice())\nlearn.fine_tune(6)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:29:37.606737Z","iopub.execute_input":"2022-09-03T19:29:37.607444Z","iopub.status.idle":"2022-09-03T19:39:39.452425Z","shell.execute_reply.started":"2022-09-03T19:29:37.607405Z","shell.execute_reply":"2022-09-03T19:39:39.451141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.model_dir='/kaggle/working/'\nlearn.save('Model1')","metadata":{"execution":{"iopub.status.busy":"2022-09-03T19:39:39.454480Z","iopub.execute_input":"2022-09-03T19:39:39.454923Z","iopub.status.idle":"2022-09-03T19:39:40.421750Z","shell.execute_reply.started":"2022-09-03T19:39:39.454872Z","shell.execute_reply":"2022-09-03T19:39:40.420831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred, _, prob = learn.predict(\"../input/hubmap-organ-segmentation/test_images/10078.tiff\")\nrle = mask2rle(pred)\nsubm1 = pd.read_csv('../input/hubmap-organ-segmentation/sample_submission.csv')\nsubm['id'] = subm1['id']\nsubm['rle'] = rle\nsubm.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T22:11:50.733792Z","iopub.execute_input":"2022-09-03T22:11:50.734366Z","iopub.status.idle":"2022-09-03T22:11:52.901964Z","shell.execute_reply.started":"2022-09-03T22:11:50.734330Z","shell.execute_reply":"2022-09-03T22:11:52.901008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_1 = tiff.imread('../input/hubmap-organ-segmentation/test_images/10078.tiff')\nmask_1 = rle2mask(subm[\"rle\"][0], (img_1.shape[1], img_1.shape[0]))\n\nplt.figure(figsize=(15,15))\nplt.subplot(1,2,1)\nplt.imshow(img_1)\n\nplt.subplot(1,2,2)\nplt.imshow(img_1)\nplt.imshow(mask_1, cmap='coolwarm', alpha=0.5)\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-09-03T20:16:30.816131Z","iopub.execute_input":"2022-09-03T20:16:30.816683Z","iopub.status.idle":"2022-09-03T20:16:33.486309Z","shell.execute_reply.started":"2022-09-03T20:16:30.816642Z","shell.execute_reply":"2022-09-03T20:16:33.485246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def make_tiles(img, tile_size=256, num_tiles=4):\n#     '''\n#     img: np.ndarray with dtype np.uint8 and shape (width, height, channel)\n#     '''\n#     w, h, ch = img.shape\n#     pad0, pad1 = (tile_size - w%tile_size) % tile_size, (tile_size - h%tile_size) % tile_size\n#     padding = [[pad0//2, pad0-pad0//2], [pad1//2, pad1-pad1//2], [0, 0]]\n#     img = np.pad(img, padding, mode='constant', constant_values=255)\n#     img = img.reshape(img.shape[0]//tile_size, tile_size, img.shape[1]//tile_size, tile_size, ch)\n#     img = img.transpose(0, 2, 1, 3, 4).reshape(-1, tile_size, tile_size, ch)\n#     if len(img) < num_tiles: # pad images so that the output shape be the same\n#         padding = [[0, num_tiles-len(img)], [0, 0], [0, 0], [0, 0]]\n#         img = np.pad(img, padding, mode='constant', constant_values=255)\n#     idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:num_tiles] # pick up Top N dark tiles\n#     img = img[idxs]\n#     return img\n","metadata":{"execution":{"iopub.status.busy":"2022-09-03T20:45:29.565205Z","iopub.execute_input":"2022-09-03T20:45:29.565656Z","iopub.status.idle":"2022-09-03T20:45:29.580607Z","shell.execute_reply.started":"2022-09-03T20:45:29.565618Z","shell.execute_reply":"2022-09-03T20:45:29.579074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def merge(mask, tile_size=256):\n#     '''\n#     img: np.ndarray with dtype np.uint8 and shape (width, height, channel)\n#     mask: np.ndarray with dtype np.uint9 and shape (width, height)\n#     '''\n#     w_i, h_i, ch = img.shape\n#     w_m, h_m     = mask.shape\n    \n#     pad0, pad1 = (tile_size - w_i%tile_size) % tile_size, (tile_size - h_i%tile_size) % tile_size\n    \n#     padding_i = [[pad0//2, pad0-pad0//2], [pad1//2, pad1-pad1//2], [0, 0]]\n#     padding_m = [[pad0//2, pad0-pad0//2], [pad1//2, pad1-pad1//2]]\n    \n#     img = np.pad(img, padding_i, mode='constant', constant_values=255)\n#     img = img.reshape(img.shape[0]//tile_size, tile_size, img.shape[1]//tile_size, tile_size, ch)\n#     img = img.transpose(0, 2, 1, 3, 4).reshape(-1, tile_size, tile_size, ch)\n    \n#     mask = np.pad(mask, padding_m, mode='constant', constant_values=255)\n#     mask = mask.reshape(mask.shape[0]//tile_size, tile_size, mask.shape[1]//tile_size, tile_size)\n#     mask = mask.transpose(0, 2, 1, 3).reshape(-1, tile_size, tile_size)\n    \n#     num_tiles = len(mask)\n#     #     if len(img) < num_tiles: # pad images so that the output shape be the same\n#     #         padding = [[0, num_tiles-len(img)], [0, 0], [0, 0], [0, 0]]\n#     #         img = np.pad(img, padding, mode='constant', constant_values=255)\n#     #idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:num_tiles] # pick up Top N dark tiles\n#     #img = img[idxs]\n#     return img, mask","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_tiles = make_tiles(img_1)\n# for i, img_crop in tqdm(enumerate(img_tiles)):\n#     pred, _, prob = learn.predict(img_crop)\n#     #reshape mask tile into one mask\n    \n#     preds.append()\n\n\n# #encode to rle\n# rle = mask2rle(preds)","metadata":{},"execution_count":null,"outputs":[]}]}