{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"height:200px;width:100%;margin: 0;\">\n    <img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/52279/logos/header.png?t=2023-05-04-23-04-20\" style=\"width:100%;\" />\n</div>","metadata":{}},{"cell_type":"code","source":"import zipfile\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nimport zipfile\nimport rasterio\nfrom rasterio.windows import Window\n\nimport json\nimport os\nimport numpy as np\nfrom tqdm import tqdm, notebook\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:14.758463Z","iopub.execute_input":"2023-05-28T22:53:14.759043Z","iopub.status.idle":"2023-05-28T22:53:15.561246Z","shell.execute_reply.started":"2023-05-28T22:53:14.759012Z","shell.execute_reply":"2023-05-28T22:53:15.560315Z"},"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 class=\"list-group-item list-group-item-action active\" data-toggle=\"list\" style=\"color:#c3448b; background:#efe9e9; border:1px dashed #efe50b;\" role=\"tab\" aria-controls=\"data\"><center>Metadata</center></h3>","metadata":{}},{"cell_type":"code","source":"tile_meta = pd.read_csv(\"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\")\nwsi_meta = pd.read_csv(\"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:15.563580Z","iopub.execute_input":"2023-05-28T22:53:15.564531Z","iopub.status.idle":"2023-05-28T22:53:15.607737Z","shell.execute_reply.started":"2023-05-28T22:53:15.564487Z","shell.execute_reply":"2023-05-28T22:53:15.606727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_meta.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:15.615669Z","iopub.execute_input":"2023-05-28T22:53:15.616374Z","iopub.status.idle":"2023-05-28T22:53:15.649356Z","shell.execute_reply.started":"2023-05-28T22:53:15.616334Z","shell.execute_reply":"2023-05-28T22:53:15.648218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:15.651345Z","iopub.execute_input":"2023-05-28T22:53:15.652180Z","iopub.status.idle":"2023-05-28T22:53:15.668702Z","shell.execute_reply.started":"2023-05-28T22:53:15.652139Z","shell.execute_reply":"2023-05-28T22:53:15.667328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 class=\"list-group-item list-group-item-action active\" data-toggle=\"list\" style=\"color:#c3448b; background:#efe9e9; border:1px dashed #efe50b;\" role=\"tab\" aria-controls=\"plotting\"><center>Plotting Data</center></h3>","metadata":{}},{"cell_type":"code","source":"tiff_file = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/0006ff2aa7cd.tif\"\ntiff_img = cv2.imread(tiff_file)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:15.670330Z","iopub.execute_input":"2023-05-28T22:53:15.671331Z","iopub.status.idle":"2023-05-28T22:53:15.723127Z","shell.execute_reply.started":"2023-05-28T22:53:15.671287Z","shell.execute_reply":"2023-05-28T22:53:15.721984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\nplt.imshow(tiff_img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:15.724581Z","iopub.execute_input":"2023-05-28T22:53:15.724912Z","iopub.status.idle":"2023-05-28T22:53:16.150260Z","shell.execute_reply.started":"2023-05-28T22:53:15.724885Z","shell.execute_reply":"2023-05-28T22:53:16.149345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 class=\"list-group-item list-group-item-action active\" data-toggle=\"list\" style=\"color:#c3448b; background:#efe9e9; border:1px dashed #efe50b;\" role=\"tab\" aria-controls=\"dataset\"><center>Dataset</center></h3>","metadata":{}},{"cell_type":"code","source":"class HubMAP_Dataset(Dataset):\n    def __init__(self, jsonPath, image_dir, augments = False, train=True):\n        # read jsonl file, \n        with open(jsonPath) as json_file:\n            json_list = list(json_file)\n        \n        # loop through the list, to create Dataframe of json Data\n        dataset = []\n        for json_str in notebook.tqdm(json_list, desc=\"Reading Json Data\"):\n            result = json.loads(json_str)\n            \n            annotations = result['annotations']\n            row = {}\n            for ann in annotations:\n                row = {}\n                row[\"id\"] = result[\"id\"]\n                row[\"type\"] = ann[\"type\"]\n                row[\"coordinates\"] = ann[\"coordinates\"]\n                row[\"mask\"] = self.coordinates_to_masks(ann[\"coordinates\"], (512, 512))[0]\n                row[\"rle\"] = self.mask2enc(row[\"mask\"])\n                dataset.append(row)\n        \n        # define dataset, to make it easier to get...\n        self.dataset = pd.DataFrame(dataset, columns=[\"id\", \"type\", \"coordinates\", \"mask\", \"rle\"])                      \n        self.train = train\n        self.image_dir = image_dir\n        \n    def enc2mask(self, encs, shape):\n        img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n        for m,enc in enumerate(encs):\n            if isinstance(enc,np.float) and np.isnan(enc): continue\n            s = enc.split()\n            for i in range(len(s)//2):\n                start = int(s[2*i]) - 1\n                length = int(s[2*i+1])\n                img[start:start+length] = 1 + m\n        return img.reshape(shape).T\n\n    def mask2enc(self, mask, n=1):\n        pixels = mask.T.flatten()\n        encs = []\n        for i in range(1,n+1):\n            p = (pixels == i).astype(np.int8)\n            if p.sum() == 0: encs.append(np.nan)\n            else:\n                p = np.concatenate([[0], p, [0]])\n                runs = np.where(p[1:] != p[:-1])[0] + 1\n                runs[1::2] -= runs[::2]\n                encs.append(' '.join(str(x) for x in runs))\n        return encs\n\n    def coordinates_to_masks(self, coordinates, shape):\n        masks = []\n        for coord in coordinates:\n            mask = np.zeros(shape, dtype=np.uint8)\n            cv2.fillPoly(mask, [np.array(coord)], 1)\n            masks.append(mask)\n        return masks\n        \n    def __getitem__(self, idx):\n        \n        data = self.dataset.iloc[idx]\n        \n        imageLoc = os.path.join(self.image_dir,data.id+\".tif\")\n        img = cv2.imread(imageLoc)\n        \n        type_struct = data.type\n        coord = data.coordinates[0]\n        \n        # create mask array\n        mask = np.zeros((512, 512), dtype=np.float32)\n        points = np.array(coord)\n        points = points.reshape((1, -1, 2))\n        mask = cv2.fillPoly(mask, pts=points, color=(255))    \n        \n        return img, type_struct, mask\n            \n    def __len__(self):\n        return len(self.dataset)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:53:20.508330Z","iopub.execute_input":"2023-05-28T22:53:20.509156Z","iopub.status.idle":"2023-05-28T22:53:20.538287Z","shell.execute_reply.started":"2023-05-28T22:53:20.509113Z","shell.execute_reply":"2023-05-28T22:53:20.537034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jsonL = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\nimage_dir = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\n\ntrainDataset = HubMAP_Dataset(jsonPath=jsonL, augments=False, image_dir=image_dir, train=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:54:20.134117Z","iopub.execute_input":"2023-05-28T22:54:20.135211Z","iopub.status.idle":"2023-05-28T22:54:50.695548Z","shell.execute_reply.started":"2023-05-28T22:54:20.135172Z","shell.execute_reply":"2023-05-28T22:54:50.694456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_masks = trainDataset.dataset\ndf_masks.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:54:50.697406Z","iopub.execute_input":"2023-05-28T22:54:50.697814Z","iopub.status.idle":"2023-05-28T22:54:51.677835Z","shell.execute_reply.started":"2023-05-28T22:54:50.697785Z","shell.execute_reply":"2023-05-28T22:54:51.676645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_masks = df_masks[df_masks['type'] == 'blood_vessel']\ndf_masks.reset_index(inplace=True, drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:55:53.788805Z","iopub.execute_input":"2023-05-28T22:55:53.789236Z","iopub.status.idle":"2023-05-28T22:55:53.804700Z","shell.execute_reply.started":"2023-05-28T22:55:53.789203Z","shell.execute_reply":"2023-05-28T22:55:53.803341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_masks['unique_id'] = df_masks['id'] + '_' + df_masks.index.astype(str)\ndf_masks.set_index('unique_id', inplace=True)\ndf_masks.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T22:55:57.346882Z","iopub.execute_input":"2023-05-28T22:55:57.347431Z","iopub.status.idle":"2023-05-28T22:55:58.364411Z","shell.execute_reply.started":"2023-05-28T22:55:57.347389Z","shell.execute_reply":"2023-05-28T22:55:58.363354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_masks.to_csv('labels.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 class=\"list-group-item list-group-item-action active\" data-toggle=\"list\" style=\"color:#c3448b; background:#efe9e9; border:1px dashed #efe50b;\" role=\"tab\" aria-controls=\"segment\"><center>Segmentation</center></h3>","metadata":{}},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"config = {\n    'resize': (768,768),\n    'resolution': (512,512),\n    'DATA': '/kaggle/input/hubmap-hacking-the-human-vasculature/train',\n    'Window' : (250,1024),\n    'bs': 64,\n    'nfolds': 4,\n    'fold': 0,\n    'NUM_WORKERS': 4,\n    'OUT_TRAIN' : 'train.zip',\n    'OUT_MASKS' : 'masks.zip'\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-28T16:29:20.550779Z","iopub.execute_input":"2023-05-28T16:29:20.551265Z","iopub.status.idle":"2023-05-28T16:29:20.558119Z","shell.execute_reply.started":"2023-05-28T16:29:20.551224Z","shell.execute_reply":"2023-05-28T16:29:20.556917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Additionals","metadata":{}},{"cell_type":"code","source":"# functions to convert encoding to mask and mask to encoding\ndef enc2mask(encs, shape):\n    '''\n    Args:\n    encs: list of rle masks\n    shape: mask shape\n    '''\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for m,enc in enumerate(encs):\n        if isinstance(enc,np.float32) and np.isnan(enc): continue\n        s = enc.split()\n        for i in range(len(s)//2):\n            start = int(s[2*i]) - 1\n            length = int(s[2*i+1])\n            img[start:start+length] = 1 + m\n    return img.reshape(shape).T\n\ndef mask2enc(mask, n=1):\n    pixels = mask.T.flatten()\n    encs = []\n    for i in range(1,n+1):\n        p = (pixels == i).astype(np.int8)\n        if p.sum() == 0: encs.append(np.nan)\n        else:\n            p = np.concatenate([[0], p, [0]])\n            runs = np.where(p[1:] != p[:-1])[0] + 1\n            runs[1::2] -= runs[::2]\n            encs.append(' '.join(str(x) for x in runs))\n    return encs","metadata":{"execution":{"iopub.status.busy":"2023-05-28T16:29:21.170584Z","iopub.execute_input":"2023-05-28T16:29:21.171009Z","iopub.status.idle":"2023-05-28T16:29:21.183376Z","shell.execute_reply.started":"2023-05-28T16:29:21.170977Z","shell.execute_reply":"2023-05-28T16:29:21.182284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sliding Windows Dataset","metadata":{}},{"cell_type":"code","source":"class HPADataset(Dataset):\n    def __init__(self, idx, resize=config['resize'], slide=config['Window'], encs=None):\n        self.data = rasterio.open(os.path.join(config['DATA'],str(idx)+'.tif'), num_threads='all_cpus')\n        # some images have issues with their format \n        # and must be saved correctly before reading with rasterio\n        if self.data.count != 3:\n            subdatasets = self.data.subdatasets\n            self.layers = []\n            if len(subdatasets) > 0:\n                for i, subdataset in enumerate(subdatasets, 0):\n                    self.layers.append(rasterio.open(subdataset))\n                    \n        self.shape = self.data.shape\n        self.slide = slide\n        self.resize = resize\n        self.idx = idx\n        self.mask = enc2mask(encs,(self.shape[1],self.shape[0])) if encs is not None else None\n        \n    def __len__(self):\n        return 1\n    \n    def __getitem__(self, idx):\n        # read img (RGB), mask (grayscale) from window slide\n        # img, mask: uint8\n        img = self.data.read([1,2,3],window=Window.from_slices(self.slide,self.slide)) \n        mask = self.mask[self.slide[0]:self.slide[1], self.slide[0]:self.slide[1]] \n        \n        # resize\n        img = cv2.resize(np.transpose(img,(1,2,0)),(self.resize[0],self.resize[1]),\n                         interpolation = cv2.INTER_AREA)\n        mask = cv2.resize(mask,(self.resize[0], self.resize[1]),\n                          interpolation = cv2.INTER_NEAREST)\n        \n        return img, mask, self.idx","metadata":{"execution":{"iopub.status.busy":"2023-05-28T16:29:21.829603Z","iopub.execute_input":"2023-05-28T16:29:21.830063Z","iopub.status.idle":"2023-05-28T16:29:21.844354Z","shell.execute_reply.started":"2023-05-28T16:29:21.830025Z","shell.execute_reply":"2023-05-28T16:29:21.843178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 class=\"list-group-item list-group-item-action active\" data-toggle=\"list\" style=\"color:#c3448b; background:#efe9e9; border:1px dashed #efe50b;\" role=\"tab\" aria-controls=\"zip\"><center>Outcome</center></h3>","metadata":{}},{"cell_type":"code","source":"x_tot,x2_tot = [],[]\nwith zipfile.ZipFile(config['OUT_TRAIN'], 'w') as img_out,\\\n zipfile.ZipFile(config['OUT_MASKS'], 'w') as mask_out:\n    for index, encs in tqdm(df_masks.iterrows(),total=len(df_masks)):\n        #image+mask dataset\n        ds = HPADataset(index.split('_')[0],encs=encs['rle'])\n        for i in range(len(ds)):\n            img, m, idx = ds[i]\n\n            x_tot.append((img/255.0).reshape(-1,3).mean(0))\n            x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0))\n\n            #write data   \n            img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n            img_out.writestr(f'{index}.png', img)\n            m = cv2.imencode('.png',m)[1]\n            mask_out.writestr(f'{index}.png', m)\n        \n#image stats\nimg_avr =  np.array(x_tot).mean(0)\nimg_std =  np.sqrt(np.array(x2_tot).mean(0) - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T16:31:13.718066Z","iopub.execute_input":"2023-05-28T16:31:13.718606Z","iopub.status.idle":"2023-05-28T16:32:51.619083Z","shell.execute_reply.started":"2023-05-28T16:31:13.718567Z","shell.execute_reply":"2023-05-28T16:32:51.615781Z"},"trusted":true},"execution_count":null,"outputs":[]}]}