{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys, os\n\nsys.path.append('../input/hubmap-submit-08') \nsys.path.append('../input/hubmap-submit-08/[third_party]')\n\n\nfrom kaggle_hubmap_kv4 import * \nimport importlib \nfrom timeit import default_timer as timer\n\nimport torch \nimport torch.cuda.amp as amp \nimport torch.nn.functional as F \nprint('import ok\\n')\n\n#-- configure --------------------------------------------- \n#image_size = 768 #512\n\norgan_threshold = { \n    'Hubmap': { \n        'kidney' : 0.28, \n        'prostate' : 0.28, \n        'largeintestine': 0.28, \n        'spleen' : 0.25, \n        'lung' : 0.08, \n    }, \n    'HPA': { \n        'kidney' : 0.50, \n        'prostate' : 0.50, \n        'largeintestine': 0.50, \n        'spleen' : 0.50, \n        'lung' : 0.10, \n    }, \n}\n\ndata_source = ['Hubmap', 'HPA'] \norgan = ['kidney', 'prostate', 'largeintestine', 'spleen', 'lung']\n\n#data_source =['Hubmap',] \n#organ = ['spleen']\n\n#submit_type = 'local-cv'\nsubmit_type = 'kaggle'\n","metadata":{"execution":{"iopub.status.busy":"2022-09-05T21:51:36.570995Z","iopub.execute_input":"2022-09-05T21:51:36.571350Z","iopub.status.idle":"2022-09-05T21:51:36.581744Z","shell.execute_reply.started":"2022-09-05T21:51:36.571321Z","shell.execute_reply":"2022-09-05T21:51:36.580649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n## model ##################################\nfrom daformer import * \nfrom smp_unet import *\n\nfrom timm.models.efficientnet import tf_efficientnet_b6\nfrom timm.models.convnext import convnext_large_384_in22ft1k\n\nfrom coat import coat_parallel_small_level5 \nfrom pvt_v2 import pvt_v2_b4, pvt_v2_b4_level5 \nfrom dualvit import *\n \n\nmodel = [ \n    dotdict( #[1]\n        is_use=1,\n        image_size=1024,\n        module='model_pvt_v2_daformer',\n        param={'encoder': pvt_v2_b4_level5, 'decoder': daformer_conv3x3, },\n        checkpoint=[\n            '../input/hubmap-submit-06-weight0/pvt_v2_b4_level5-daformer_conv3x3-aug5b-1024-fold-3-swa.pth',\n            '../input/hubmap-submit-06-weight0/pvt_v2_b4_level5-daformer_conv3x3-aug5b-1024-fold-2-swa.pth', \n            '../input/hubmap-submit-06-weight0/pvt_v2_b4_level5-daformer_conv3x3-aug5b-1024-fold-1-swa.pth',\n            \n            #'../input/hubmap-submit-06-weight0/pvt_v2_b4_level5-daformer_conv3x3-aug5b-1024-fold-4-swa.pth',\n             \n        ],\n    ),\n    dotdict(#[2]\n        is_use = 1,\n        image_size = 1536,\n        module = 'model_coat_daformer',\n        param={'encoder': coat_parallel_small_level5, 'decoder':daformer_conv1x1},\n        checkpoint = [ \n            '../input/hubmap-submit-06-weight0/coat_parallel_small_level5-daformer_conv1x1-aug8b-1536-fold-3-swa.pth',\n            '../input/hubmap-submit-06-weight0/coat_parallel_small_level5-daformer_conv1x1-aug8b-1536-fold-2-swa.pth', \n            '../input/hubmap-submit-06-weight0/coat_parallel_small_level5-daformer_conv1x1-aug8b-1536-fold-1-swa.pth',\n            \n            #'../input/hubmap-submit-06-weight0/coat_parallel_small_level5-daformer_conv1x1-aug8b-1536-fold-4-swa.pth', \n            #'../input/hubmap-submit-06-weight0/coat_parallel_small_level5-daformer_conv1x1-aug8b-1536-fold-0-swa.pth',  \n            '../input/hubmap-submit-06-weight1/coat_parallel_small_level5-daformer_conv1x1-aug8b-diff-seed-1536-fold-1-best-swa.pth', #0.80\n            '../input/hubmap-submit-06-weight1/coat_parallel_small_level5-daformer_conv1x1-aug8b-diff-seed-1536-fold-4-best-swa.pth',#0.80\n        \n        ],\n    ),\n \n    \n    dotdict( #[3]\n        is_use=1,\n        image_size=1536,\n        module='model_effnet_smp_unet',\n        param={'encoder': tf_efficientnet_b6, 'decoder': smp_unet},  \n        checkpoint=[\n            '../input/hubmap-submit-06-weight0/tf_efficientnet_b6-unet-aug8b-1536-fold-3-swa.pth', \n            '../input/hubmap-submit-06-weight0/tf_efficientnet_b6-unet-aug8b-1536-fold-2-swa.pth',\n            '../input/hubmap-submit-06-weight0/tf_efficientnet_b6-unet-aug8b-1536-fold-1-swa.pth',\n        ],\n    ),\n    dotdict(#[4]\n        is_use = 1,\n        image_size = 768,\n        module = 'model_dualvit_daformer',\n        param={'encoder': dual_vit_b, 'decoder':daformer_conv3x3},\n        checkpoint = [\n             '../input/hubmap-submit-06-weight0/dual_vit_b-daformer_conv3x3-aug8-768-fold-3-swa.pth',\n             '../input/hubmap-submit-06-weight0/dual_vit_b-daformer_conv3x3-aug8-768-fold-2-swa.pth', \n             '../input/hubmap-submit-06-weight0/dual_vit_b-daformer_conv3x3-aug8-768-fold-1-swa.pth',\n             \n             #'../input/hubmap-submit-06-weight0/dual_vit_b-daformer_conv3x3-aug8-768-fold-4-swa.pth',\n        ],\n    ), \n\n    \n    dotdict(#[5]\n        is_use=1,\n        image_size=1280,\n        module='model_convnext_smp_unet',\n        param={'encoder': convnext_large_384_in22ft1k, 'decoder': smp_unet},   \n        checkpoint=[\n            '../input/hubmap-submit-06-weight0/convnext_large_384_in22ft1k-smp_unet-aug8b-1280-fold-3-swa.pth',\n            '../input/hubmap-submit-06-weight0/convnext_large_384_in22ft1k-smp_unet-aug8b-1280-fold-2-swa.pth',\n            '../input/hubmap-submit-06-weight0/convnext_large_384_in22ft1k-smp_unet-aug8b-1280-fold-1-swa.pth',\n            \n            #'../input/hubmap-submit-06-weight0/convnext_large_384_in22ft1k-smp_unet-aug8b-1280-fold-4-swa.pth',\n        ],\n    ), \n    \n    \n#     dotdict(\n#         is_use=0,  # 0,#1,\n#         image_size=768,\n#         module='model_pvt_v2_smp_unet',\n#         param={'encoder': pvt_v2_b4, 'decoder': None},\n#         checkpoint=[\n#             '../input/hubmap-submit-06-weight0/pvt_v2_b4-smp_unet-aug8b-768-fold-3-swa.pth',\n\n#         ],\n#     ),\n \n      dotdict(\n        is_use=1,\n        image_size=768,\n        module='model_pvt_v2_daformer',\n        param={'encoder': pvt_v2_b4, 'decoder': daformer_conv3x3},   \n        checkpoint=[\n            #'../input/hubmap-submit-06-weight1/pvt_v2_b4-aug8b-768-different-seed-fold-0-swa.pth', #lb=0.78\n            #'../input/hubmap-submit-06-weight1/pvt_v2_b4-aug8b-768-different-seed-fold-2-swa.pth', #0.79 \n            #'../input/hubmap-submit-06-weight1/pvt_v2_b4-aug8b-768-different-seed-fold-3-swa.pth',  #0.75\n            '../input/hubmap-submit-06-weight1/pvt_v2_b4-aug8b-768-different-seed-fold-1-swa.pth', #0.80\n            '../input/hubmap-submit-06-weight1/pvt_v2_b4-aug8b-768-different-seed-fold-4-swa.pth',  #0.80\n        ],\n    ),\n] ","metadata":{"execution":{"iopub.status.busy":"2022-09-05T21:51:36.584424Z","iopub.execute_input":"2022-09-05T21:51:36.585392Z","iopub.status.idle":"2022-09-05T21:51:36.603575Z","shell.execute_reply.started":"2022-09-05T21:51:36.585352Z","shell.execute_reply":"2022-09-05T21:51:36.602642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## dataset #####\n\nif submit_type == 'local-cv':\n    valid_file = '../input/hubmap-submit-08/valid_df.fold0.csv'\n    valid_file = '../input/hubmap-submit-08/valid_df.fold1.diff-seed.csv'\n    tiff_dir   = '../input/hubmap-organ-segmentation/train_images'\n\nif submit_type == 'kaggle':\n    valid_file = '../input/hubmap-organ-segmentation/test.csv'\n    tiff_dir   = '../input/hubmap-organ-segmentation/test_images'\n\nvalid_df = pd.read_csv(valid_file)\nvalid_df.loc[:,'img_area']=valid_df['img_height']*valid_df['img_width']#sort by biggest image first for memory debug\nvalid_df = valid_df.sort_values('img_area').reset_index(drop=True)\nprint('load valid_df ok')\n\n\ndef image_to_tensor(image, mode='rgb'):\n    if  mode=='bgr' :\n        image = image[:,:,::-1]\n    \n    x = image.transpose(2,0,1)\n    x = np.ascontiguousarray(x)\n    x = torch.tensor(x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-09-05T21:51:36.608526Z","iopub.execute_input":"2022-09-05T21:51:36.608862Z","iopub.status.idle":"2022-09-05T21:51:36.628254Z","shell.execute_reply.started":"2022-09-05T21:51:36.608835Z","shell.execute_reply":"2022-09-05T21:51:36.626959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## submission ##\n\ndef do_local_validation():\n    print('\\tlocal validation ...')\n    \n    submit_df = pd.read_csv('submission.csv').fillna('')\n    submit_df = submit_df.sort_values('id')\n    truth_df  = valid_df.sort_values('id')\n    \n    lb_score = []\n    num = len(submit_df)\n    for i in range(num):\n        t_df = truth_df.iloc[i]\n        p_df = submit_df.iloc[i]\n        t = rle_decode(t_df.rle, t_df.img_height, t_df.img_width, 1)\n        p = rle_decode(p_df.rle, t_df.img_height, t_df.img_width, 1)\n        \n        dice = 2*(t*p).sum()/(p.sum()+t.sum())\n        lb_score.append(dice)\n        \n        if 0:\n            overlay = result_to_overlay(p, t)\n            image_show_norm('overlay', overlay, min=0, max=1, resize=0.10)\n            cv2.waitKey(1)\n\n    truth_df.loc[:,'lb_score']=lb_score\n    for organ in ['all', 'kidney', 'prostate', 'largeintestine', 'spleen', 'lung']:\n        if organ != 'all':\n            d = truth_df[truth_df.organ == organ]\n        else:\n            d = truth_df\n        print('\\t%f\\t%s\\t%f' % (len(d) / len(truth_df), organ, d.lb_score.mean()))\n        \n    \ndef load_net(model):\n    print('\\tload %s ... '%(model.module))\n    M = importlib.import_module(model.module)\n    num = len(model.checkpoint)\n    net = []\n    for f in range(num):\n        print('\\t\\t%s' % (model.checkpoint[f]))\n        n = M.Net(**model.param)\n        s = torch.load(model.checkpoint[f], map_location=lambda storage, loc: storage) ['state_dict']\n        ret = n.load_state_dict(s, strict=False)\n\n        missing_keys = ret.missing_keys\n        if model.param['encoder']==coat_parallel_small_level5:\n            missing_keys = [m for m in missing_keys if not any(s in m for s in [\n                '.mlp2.','.mlp3.','.mlp4.','.mlp5.','.cpe.','.crpe.'])]\n        print('\\t\\tmissing_keys=%s' % str(missing_keys))\n\n        n.cuda()\n        n.eval()\n        net.append(n)\n        \n    print('ok!')\n    return net\n\n\ndef do_tta_batch(image, organ):\n    \n    batch = { #<todo> multiscale????\n        'image': torch.stack([\n            image,\n            torch.flip(image,dims=[1]),\n            torch.flip(image,dims=[2]),\n        ]),\n        'organ': torch.Tensor(\n            [[organ_to_label[organ]]]*3\n        ).long()\n    }\n    return batch\n\ndef undo_tta_batch(probability):\n    probability[0] = probability[0]\n    probability[1] = torch.flip(probability[1],dims=[1])\n    probability[2] = torch.flip(probability[2],dims=[2])\n    probability = probability.mean(0, keepdims=True)\n    probability = probability[0,0].float()\n    return probability\n\ndef do_submit(): \n    print('** submit_type  = %s *******************'%submit_type)\n \n    for m in model:\n        if m.is_use==1:\n            m.net = load_net(m)\n        else:\n            m.net = None\n    \n    result = []\n    start_timer = timer()\n    for i,d in valid_df.iterrows():\n        id = d['id']\n        if (d['data_source'] in data_source) and (d['organ'] in organ):\n            \n            tiff_file = tiff_dir +'/%d.tiff'%id\n            tiff = read_tiff(tiff_file, 'rgb')  \n            tiff = tiff.astype(np.float32)/255\n            H,W,_ = tiff.shape\n            \n            use = 0\n            probability = 0\n            for image_size in [768, 1024, 1280, 1536]:\n                if 0: \n                    s = d.pixel_size/0.4 * (image_size/3000)\n                    h = int(np.ceil(int(H*s)/32)*32)\n                    w = int(np.ceil(int(W*s)/32)*32)\n                    image = cv2.resize(tiff,dsize=(w,h),interpolation=cv2.INTER_LINEAR)\n                else:\n                    #image = cv2.resize(image,dsize=(800,800),interpolation=cv2.INTER_LINEAR)\n                    #image = cv2.resize(image,dsize=(768,768),interpolation=cv2.INTER_LINEAR)\n                    image = cv2.resize(tiff,dsize=(image_size,image_size),interpolation=cv2.INTER_LINEAR)\n                \n             \n                image = image_to_tensor(image, mode='rbg')\n                batch = { k:v.cuda() for k,v in do_tta_batch(image, d.organ).items() }\n        \n                with torch.no_grad():\n                    with amp.autocast(enabled = True):\n                        \n                        for m in model:\n                            if m.net is  None: continue\n                            if m.image_size != image_size: continue\n                            \n                            for n in m.net:\n                                #print('running...',m.module, image_size)\n                                use += 1\n                                output = n(batch)#data_parallel(net, batch) #\n                                prob = F.interpolate(output['probability'], size=(H,W), #size=(d.img_height,d.img_width),\n                                                  mode='bilinear',align_corners=False, antialias=True )\n                                probability += prob\n     \n            #---\n            probability = undo_tta_batch(probability/use)\n            probability = probability.data.cpu().numpy()\n            p = probability>organ_threshold[d.data_source][d.organ] \n            #print('p.sum()', p.sum())\n            if p.sum()==0:   #at least one FTU\n                p = np.zeros((H,W), np.float32)\n                cv2.circle(p,(W//2,H//2),int(W*0.4), 1, -1)\n                p = p>0.5 \n            rle = rle_encode(p) \n        else:\n            rle = ''\n        \n        #----\n        if 0: #debug\n            image = cv2.cvtColor(tiff, 4).astype(np.float32)/255 #cv2.COLOR_RGB2BGR=4\n            mask  = rle_decode(d.rle, d.img_height, d.img_width, 1) #None\n            overlay = result_to_overlay(image, mask, probability)\n            \n            #image_show('image',image, resize=0.25)\n            image_show('overlay',overlay, resize=0.25)\n            cv2.waitKey(0)\n            pass\n        \n        result.append({ 'id':id, 'rle':rle, })\n        print('\\r', '\\tsubmit ... %3d/%3d %s'%(i, len(valid_df), time_to_str(timer() - start_timer,'sec')), end='',flush=True)\n        torch.cuda.empty_cache()\n        \n    print('\\n')\n    \n    #---\n    submit_df = pd.DataFrame(result)\n    submit_df.to_csv('submission.csv',index=False)\n    print(submit_df)\n    print('\\tsubmit_df ok!')\n    print('')\n    \n    if submit_type  == 'local-cv':\n        do_local_validation()\n        \n    if len(submit_df)==1:\n        import matplotlib.pyplot as plt \n        m = tiff\n        b = probability\n        p = p.astype(np.float32)\n        \n        plt.figure(figsize=(16, 7))\n        plt.subplot(1, 4, 1); plt.imshow(m); plt.axis('OFF'); plt.title('image')\n        plt.subplot(1, 4, 2); plt.imshow(b*255); plt.axis('OFF'); plt.title('probability')\n        plt.subplot(1, 4, 3); plt.imshow(m); plt.imshow(b*255, alpha=0.4); plt.axis('OFF'); plt.title('image+probability')\n        plt.subplot(1, 4, 4); plt.imshow(p*255); plt.axis('OFF'); plt.title('predict')\n        plt.tight_layout()\n        plt.show()\ndo_submit()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T21:51:36.825163Z","iopub.execute_input":"2022-09-05T21:51:36.825555Z","iopub.status.idle":"2022-09-05T21:53:37.774501Z","shell.execute_reply.started":"2022-09-05T21:51:36.825523Z","shell.execute_reply":"2022-09-05T21:53:37.773635Z"},"trusted":true},"execution_count":null,"outputs":[]}]}