{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## <center></center>\n# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">👋🏼 Inference notebook may be late, but it will never be absent!</p></div>\n","metadata":{}},{"cell_type":"markdown","source":"#### I've got a score of 0.78 on LB by the same notebook yesterday.\n\n* Single fold out of 5 folds\n* No external data\n* No post process(even no specific threshold for each class)\n* LB: 0.78 Local: 0.818\n\n#### What have I done?\n\n* Tiles with large size\n* Multi-class segmentation\n* Large model with some modifications\n* Kind of good training strategy \n---","metadata":{}},{"cell_type":"markdown","source":"### *My related works*\n\n1. Data-Prepareing\n> [Multi-class dataset  (notebook)](https://www.kaggle.com/code/w3579628328/6-classes-dataset-for-mmsegmentation)\n2. Dataset (already created)\n* > [256x256  (dataset)](https://www.kaggle.com/datasets/w3579628328/mmsegmentation256x256)\n* > [512x512  (dataset)](https://www.kaggle.com/datasets/w3579628328/mmsegmentation512x512)\n3. Training (Multi-class)\n> [Multi-class training with mmsegmentation](https://www.kaggle.com/code/w3579628328/multi-class-mmsegmentation-training)\n4. Training (Single-class)\n> [Binary class training with mmsegmentation](https://www.kaggle.com/code/w3579628328/mmsegmentation-trainning)\n\n\n#### enjoy🤗🤗🤗\n---","metadata":{}},{"cell_type":"markdown","source":"#### Dependents\n","metadata":{"papermill":{"duration":0.00991,"end_time":"2021-03-12T06:33:14.88117","exception":false,"start_time":"2021-03-12T06:33:14.87126","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install ../input/mmdetection/addict-2.4.0-py3-none-any.whl\n!pip install ../input/mmdetection/yapf-0.31.0-py2.py3-none-any.whl\n!pip install ../input/mmdetection/terminaltables-3.1.0-py3-none-any.whl\n!pip install ../input/mmdetection/einops-0.4.1-py3-none-any.whl\n!pip install ../input/mmsegmentation/mmcv-full/mmcv_full-1.5.3-cp37-cp37m-linux_x86_64.whl\n!pip install ../input/openmmlab-essential-repositories/openmmlab-repos/src/mmcls-0.23.1-py2.py3-none-any.whl","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-06T12:28:32.827071Z","iopub.execute_input":"2022-08-06T12:28:32.827780Z","iopub.status.idle":"2022-08-06T12:32:40.357254Z","shell.execute_reply.started":"2022-08-06T12:28:32.827640Z","shell.execute_reply":"2022-08-06T12:32:40.355413Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r ../input/mmsegm/mmsegmentation-master /kaggle/working/ && cd /kaggle/working/mmsegmentation-master && pip install -e . && cd ..","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:32:40.372898Z","iopub.execute_input":"2022-08-06T12:32:40.380090Z","iopub.status.idle":"2022-08-06T12:33:36.413693Z","shell.execute_reply.started":"2022-08-06T12:32:40.380039Z","shell.execute_reply":"2022-08-06T12:33:36.411908Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Import librarys","metadata":{}},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport pandas as pd\nimport os\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport sys\nimport gc\nsys.path.append('./mmsegmentation-master')\nfrom mmseg.apis import init_segmentor, inference_segmentor\nfrom mmcv.utils import config","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":3.066435,"end_time":"2021-03-12T06:33:17.956368","exception":false,"start_time":"2021-03-12T06:33:14.889933","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T12:40:39.574640Z","iopub.execute_input":"2022-08-06T12:40:39.575756Z","iopub.status.idle":"2022-08-06T12:40:39.583417Z","shell.execute_reply.started":"2022-08-06T12:40:39.575724Z","shell.execute_reply":"2022-08-06T12:40:39.581646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load models","metadata":{}},{"cell_type":"code","source":"configs = [\n    '../input/my-model-and-cfg/my_config.py',\n]\nckpts = [\n    '../input/my-model-and-cfg/self_model_fold4.pth',\n]\n\nDATA = '../input/hubmap-organ-segmentation/test_images/'\ndf_sample = pd.read_csv('../input/hubmap-organ-segmentation/sample_submission.csv').set_index('id')\n\nmodels = []\nfor idx,(cfg, ckpt) in enumerate(zip(configs, ckpts)):\n    cfg = config.Config.fromfile(cfg)\n    model = init_segmentor(cfg, ckpt, device='cuda:0')\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:40:41.090249Z","iopub.execute_input":"2022-08-06T12:40:41.090716Z","iopub.status.idle":"2022-08-06T12:40:42.421760Z","shell.execute_reply.started":"2022-08-06T12:40:41.090686Z","shell.execute_reply":"2022-08-06T12:40:42.420412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### mask -> rle","metadata":{"papermill":{"duration":0.008314,"end_time":"2021-03-12T06:33:18.008555","exception":false,"start_time":"2021-03-12T06:33:18.000241","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def rle_encode_less_memory(img):\n    pixels = img.T.flatten()\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"_kg_hide-input":false,"papermill":{"duration":0.026676,"end_time":"2021-03-12T06:33:18.043775","exception":false,"start_time":"2021-03-12T06:33:18.017099","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T12:40:42.424554Z","iopub.execute_input":"2022-08-06T12:40:42.425765Z","iopub.status.idle":"2022-08-06T12:40:42.434654Z","shell.execute_reply.started":"2022-08-06T12:40:42.425718Z","shell.execute_reply":"2022-08-06T12:40:42.433097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Inference section from Fastai-base, which is more flexible than inference-api from mmsegmentation, you may get a better score with proper modification.","metadata":{}},{"cell_type":"markdown","source":"sz = 256\nreduce = 4\nmean_std = np.load('/kaggle/input/stain-dataset-256x256-macenko/avr_std.npz')\nmean = mean_std['mean']\nstd = mean_std['std']\ns_th = 40\np_th = 1000*(sz//256)**2\nidentity = rasterio.Affine(1, 0, 0, 0, 1, 0)\n\ndef img2tensor(img,dtype:np.dtype=np.float32):\n    if img.ndim==2 : img = np.expand_dims(img,2)\n    img = np.transpose(img,(2,0,1))\n    return torch.from_numpy(img.astype(dtype, copy=False))\n\nclass HuBMAPDataset(Dataset):\n    def __init__(self, idx, sz=sz, reduce=reduce):\n        self.data = rasterio.open(os.path.join(DATA,idx+'.tiff'), transform = identity,\n                                 num_threads='all_cpus')\n        # some images have issues with their format \n        # and must be saved correctly before reading with rasterio\n        if self.data.count != 3:\n            subdatasets = self.data.subdatasets\n            self.layers = []\n            if len(subdatasets) > 0:\n                for i, subdataset in enumerate(subdatasets, 0):\n                    self.layers.append(rasterio.open(subdataset))\n        self.shape = self.data.shape\n        self.reduce = reduce\n        self.sz = reduce*sz\n        self.pad0 = (self.sz - self.shape[0]%self.sz)%self.sz\n        self.pad1 = (self.sz - self.shape[1]%self.sz)%self.sz\n        self.n0max = (self.shape[0] + self.pad0)//self.sz\n        self.n1max = (self.shape[1] + self.pad1)//self.sz\n        if idx == '10078':\n            self.apply_stain_norm = False\n        else:\n            self.apply_stain_norm = True\n        \n    def __len__(self):\n        return self.n0max*self.n1max\n    \n    def __getitem__(self, idx):\n        n0,n1 = idx//self.n1max, idx%self.n1max\n        x0,y0 = -self.pad0//2 + n0*self.sz, -self.pad1//2 + n1*self.sz\n        # make sure that the region to read is within the image\n        p00,p01 = max(0,x0), min(x0+self.sz,self.shape[0])\n        p10,p11 = max(0,y0), min(y0+self.sz,self.shape[1])\n        img = np.zeros((self.sz,self.sz,3),np.uint8)\n        # mapping the loade region to the tile\n        if self.data.count == 3:\n            img[(p00-x0):(p01-x0),(p10-y0):(p11-y0)] = np.moveaxis(self.data.read([1,2,3],\n                window=Window.from_slices((p00,p01),(p10,p11))), 0, -1)\n        else:\n            for i,layer in enumerate(self.layers):\n                img[(p00-x0):(p01-x0),(p10-y0):(p11-y0),i] =\\\n                  layer.read(1,window=Window.from_slices((p00,p01),(p10,p11)))\n        \n        if self.reduce != 1:\n            img = cv2.resize(img,(self.sz//reduce,self.sz//reduce),\n                             interpolation = cv2.INTER_AREA)\n        \n        if self.apply_stain_norm == True:  \n            img = T(img)\n            img, H, E = torch_normalizer.normalize(I=img, stains=True)\n            img = img.numpy().astype(np.uint8)\n#             img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        #check for empty imges\n        hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n        h,s,v = cv2.split(hsv)\n        \n        if (s>s_th).sum() <= p_th or img.sum() <= p_th:\n            #images with -1 will be skipped\n            return img2tensor((img/255.0 - mean)/std), -1\n        else: return img2tensor((img/255.0 - mean)/std), idx","metadata":{"papermill":{"duration":0.037945,"end_time":"2021-03-12T06:33:18.090462","exception":false,"start_time":"2021-03-12T06:33:18.052517","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T12:36:08.437913Z","iopub.execute_input":"2022-08-06T12:36:08.439286Z","iopub.status.idle":"2022-08-06T12:36:08.506605Z","shell.execute_reply.started":"2022-08-06T12:36:08.439227Z","shell.execute_reply":"2022-08-06T12:36:08.503934Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}}},{"cell_type":"markdown","source":"#iterator like wrapper that returns predicted masks\nclass Model_pred:\n    def __init__(self, models, dl, tta:bool=True, half:bool=False):\n        self.models = models\n        self.dl = dl\n        self.tta = tta\n        self.half = half\n        \n    def __iter__(self):\n        count=0\n        with torch.no_grad():\n            for x,y in iter(self.dl):\n                if ((y>=0).sum() > 0): #exclude empty images\n                    x = x[y>=0].to(device)\n                    y = y[y>=0]\n                    if self.half: x = x.half()\n                    py = None\n                    for model in self.models:\n                        p = model(x)\n                        p = torch.sigmoid(p).detach()\n                        if py is None: py = p\n                        else: py += p\n                    if self.tta:\n                        #x,y,xy flips as TTA\n                        flips = [[-1],[-2],[-2,-1]]\n                        for f in flips:\n                            xf = torch.flip(x,f)\n                            for model in self.models:\n                                p = model(xf)\n                                p = torch.flip(p,f)\n                                py += torch.sigmoid(p).detach()\n                        py /= (1+len(flips))        \n                    py /= len(self.models)\n\n                    py = F.upsample(py, scale_factor=reduce, mode=\"bilinear\")\n                    py = py.permute(0,2,3,1).float().cpu()\n                    \n                    batch_size = len(py)\n                    for i in range(batch_size):\n                        yield py[i],y[i]\n                        count += 1\n                    \n    def __len__(self):\n        return len(self.dl.dataset)","metadata":{"papermill":{"duration":0.030274,"end_time":"2021-03-12T06:33:18.129546","exception":false,"start_time":"2021-03-12T06:33:18.099272","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T12:35:29.609910Z","iopub.execute_input":"2022-08-06T12:35:29.611269Z","iopub.status.idle":"2022-08-06T12:35:29.632736Z","shell.execute_reply.started":"2022-08-06T12:35:29.611219Z","shell.execute_reply":"2022-08-06T12:35:29.631090Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}}},{"cell_type":"markdown","source":"def fastai_inf():\n    names,preds = [],[]\n    for idx,row in tqdm(df_sample.iterrows(),total=len(df_sample)):\n        idx = str(row['id'])\n        ds = HuBMAPDataset(idx)\n        dl = DataLoader(ds,bs,num_workers=0,shuffle=False,pin_memory=True)\n        mp = Model_pred(models,dl)\n        mask = torch.zeros(len(ds),ds.sz,ds.sz,dtype=torch.int8)\n        for p,i in iter(mp): mask[i.item()] = p.squeeze(-1) > TH\n        mask = mask.view(ds.n0max,ds.n1max,ds.sz,ds.sz).\\\n            permute(0,2,1,3).reshape(ds.n0max*ds.sz,ds.n1max*ds.sz)\n        mask = mask[ds.pad0//2:-(ds.pad0-ds.pad0//2) if ds.pad0 > 0 else ds.n0max*ds.sz,\n            ds.pad1//2:-(ds.pad1-ds.pad1//2) if ds.pad1 > 0 else ds.n1max*ds.sz]\n        rle = rle_encode_less_memory(mask.numpy())\n        names.append(idx)\n        preds.append(rle)\n        del mask, ds, dl\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:35:29.635936Z","iopub.execute_input":"2022-08-06T12:35:29.639299Z","iopub.status.idle":"2022-08-06T12:35:29.658198Z","shell.execute_reply.started":"2022-08-06T12:35:29.639229Z","shell.execute_reply":"2022-08-06T12:35:29.656304Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}}},{"cell_type":"markdown","source":"#### My inference","metadata":{"_kg_hide-output":true,"papermill":{"duration":638.710533,"end_time":"2021-03-12T06:44:10.832427","exception":false,"start_time":"2021-03-12T06:33:32.121894","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-23T00:49:52.322916Z","iopub.execute_input":"2022-06-23T00:49:52.323682Z","iopub.status.idle":"2022-06-23T00:49:53.476194Z","shell.execute_reply.started":"2022-06-23T00:49:52.323641Z","shell.execute_reply":"2022-06-23T00:49:53.475228Z"}}},{"cell_type":"code","source":"'''\nTo ensemble models, you need to do some modifications with \n\"../input/mmsegm/mmsegmentation-master/mmseg/models/segmentors/encoder_decoder.py\"\n'''\nnames,preds = [],[]\nimgs, pd_mks = [],[]\ndebug = len(df_sample)<2\nfor idx,row in tqdm(df_sample.iterrows(),total=len(df_sample)):\n    img  = cv2.imread(os.path.join(DATA,str(idx)+'.tiff'))\n    im_  = img/img.max()\n    pred = inference_segmentor(models[0], im_)[0]\n    pred = (pred>0).astype(np.uint8)\n    rle = rle_encode_less_memory(pred)\n    names.append(str(idx))\n    preds.append(rle)\n    if debug:\n        imgs.append(img)\n        pd_mks.append(pred)\n    del img, pred, rle, idx, row\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:41:31.936933Z","iopub.execute_input":"2022-08-06T12:41:31.938648Z","iopub.status.idle":"2022-08-06T12:41:48.044765Z","shell.execute_reply.started":"2022-08-06T12:41:31.938582Z","shell.execute_reply":"2022-08-06T12:41:48.043175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if debug:\n    import matplotlib.pyplot as plt\n    for img, mask in zip(imgs, pd_mks):\n        plt.figure(figsize=(12, 7))\n        plt.subplot(1, 3, 1); plt.imshow(img); plt.axis('OFF'); plt.title('image')\n        plt.subplot(1, 3, 2); plt.imshow(mask*255); plt.axis('OFF'); plt.title('mask')\n        plt.subplot(1, 3, 3); plt.imshow(img); plt.imshow(mask*255, alpha=0.4); plt.axis('OFF'); plt.title('overlay')\n        plt.tight_layout()\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:41:48.047598Z","iopub.execute_input":"2022-08-06T12:41:48.048427Z","iopub.status.idle":"2022-08-06T12:41:49.984093Z","shell.execute_reply.started":"2022-08-06T12:41:48.048384Z","shell.execute_reply":"2022-08-06T12:41:49.982924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:42:45.746183Z","iopub.execute_input":"2022-08-06T12:42:45.747244Z","iopub.status.idle":"2022-08-06T12:42:47.386451Z","shell.execute_reply.started":"2022-08-06T12:42:45.747194Z","shell.execute_reply":"2022-08-06T12:42:47.384596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'id':names,'rle':preds})\ndf.to_csv('submission.csv',index=False)","metadata":{"papermill":{"duration":0.419953,"end_time":"2021-03-12T06:44:11.262501","exception":false,"start_time":"2021-03-12T06:44:10.842548","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T12:42:47.390510Z","iopub.execute_input":"2022-08-06T12:42:47.392089Z","iopub.status.idle":"2022-08-06T12:42:47.407061Z","shell.execute_reply.started":"2022-08-06T12:42:47.392035Z","shell.execute_reply":"2022-08-06T12:42:47.405324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.010126,"end_time":"2021-03-12T06:44:11.283196","exception":false,"start_time":"2021-03-12T06:44:11.27307","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}