{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"},{"sourceId":7002347,"sourceType":"datasetVersion","datasetId":4025450},{"sourceId":7002447,"sourceType":"datasetVersion","datasetId":4025494}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys, os\nsys.path.append('/kaggle/input/blood-vessel-segmentation-third-party')\nsys.path.append('/kaggle/input/blood-vessel-segmentation-00')\n\nfrom helper import *\n\nimport cv2\nimport pandas as pd\nfrom glob import glob\nimport numpy as np\n\nfrom timeit import default_timer as timer\n\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nprint('IMPORT OK  !!!!')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-21T06:15:23.239334Z","iopub.execute_input":"2023-11-21T06:15:23.239702Z","iopub.status.idle":"2023-11-21T06:15:23.246530Z","shell.execute_reply.started":"2023-11-21T06:15:23.239677Z","shell.execute_reply":"2023-11-21T06:15:23.245520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = dotdict(\n    batch_size = 3,\n)\n\nmode = 'submit' # 'local' #\n\ndata_dir = \\\n    '/kaggle/input/blood-vessel-segmentation'\n\n\nif 'local' in mode:\n    valid_folder = [\n        ('kidney_1_dense', (450, 500)),\n        ('kidney_3_sparse',(450, 500)),\n    ] #debug for local development\n    \n    valid_file = []\n    for image_folder, image_no in valid_folder:\n        f = [f'{data_dir}/train/{image_folder}/images/{i:04d}.tif' for i in range(*image_no)]\n        f = sorted(f)\n        valid_file.append(f)\n        \nif 'submit' in mode:\n    valid_file = []\n    valid_folder = sorted(glob(f'{data_dir}/test/*'))\n    for image_folder in valid_folder:\n        f = sorted(glob(f'{image_folder}/images/*.tif'))\n        valid_file.append(f)\n\n    glob_file = glob(f'{data_dir}/kidney_5/images/*.tif')\n    if len(glob_file)==3:\n        mode = 'submit-fake' #fake submission to save gpu time when submitting\n        #todo .....\n\n\n\n\nprint('len(valid_file) :', len(valid_file))\nprint(valid_file[0][:3])\n#print(valid_file[1])\n#print(valid_file[-1])\nprint('')\n\n\nprint('MODE OK  !!!!')","metadata":{"execution":{"iopub.status.busy":"2023-11-21T06:15:23.248157Z","iopub.execute_input":"2023-11-21T06:15:23.248477Z","iopub.status.idle":"2023-11-21T06:15:23.317826Z","shell.execute_reply.started":"2023-11-21T06:15:23.248429Z","shell.execute_reply":"2023-11-21T06:15:23.316871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def file_to_id(f):\n    s = f.split('/')\n    return s[-3]+'_' + s[-1][:-4]\n\nclass MyLoader(object):\n    def __init__(self,):\n        self.split = []\n        for v in valid_file:\n            self.split  += np.array_split(v, max(1,int(len(v)//cfg.batch_size)))\n     \n    def __len__(self,):\n        return len(self.split)\n\n    def __getitem__(self, index):\n        file = self.split[index]\n\n        image = []\n        for f in file:\n           \n            m = cv2.imread(f,cv2.IMREAD_GRAYSCALE)\n\n            #---\n            #process image\n            m = (m - m.min())/(m.max() - m.min() + 0.0001)\n\n            #---\n            image.append(m)\n            \n        image = np.stack(image)\n        image = torch.from_numpy(image).float().unsqueeze(1)\n        return dotdict(\n            id=[file_to_id(f) for f in file],\n            image=image,\n        )\n\nprint('DATASET OK  !!!!')","metadata":{"execution":{"iopub.status.busy":"2023-11-21T06:15:23.319892Z","iopub.execute_input":"2023-11-21T06:15:23.320613Z","iopub.status.idle":"2023-11-21T06:15:23.330491Z","shell.execute_reply.started":"2023-11-21T06:15:23.320577Z","shell.execute_reply":"2023-11-21T06:15:23.329491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/456033\ndef remove_small_objects(mask, min_size):\n    # Find all connected components (labels)\n    num_label, label, stats, centroid = cv2.connectedComponentsWithStats(mask, connectivity=8)\n\n    # create a mask where small objects are removed\n    processed = np.zeros_like(mask)\n    for l in range(1, num_label):\n        if stats[l, cv2.CC_STAT_AREA] >= min_size:\n            processed[label == l] = 255\n\n    return processed\n\ndef rle_encode(mask):\n    pixel = mask.flatten()\n    pixel = np.concatenate([[0], pixel, [0]])\n    run = np.where(pixel[1:] != pixel[:-1])[0] + 1\n    run[1::2] -= run[::2]\n    rle = ' '.join(str(r) for r in run)\n    if rle == '':\n        rle = '1 0'\n    return rle\n\n#-------------------------------\n\n\nfrom model import *\ncheckpoint_file = \\\n    '/kaggle/input/blood-vessel-segmentation-00/00001707.pth'\n\nnet = Net()\n#run_check_net()\nstate_dict = torch.load(checkpoint_file, map_location=lambda storage, loc: storage)['state_dict']\nprint(net.load_state_dict(state_dict, strict=False))  # True\n\nnet = net.eval()\nnet = net.cuda()\n#net = torch.compile(net)\n\ndef do_submit():\n    \n    valid_loader = MyLoader()\n    total_num_batch = len(valid_loader)\n    \n    df_data = []\n    start_timer = timer()\n    for t in range(total_num_batch):\n        print(f'\\r validation: {t}/{total_num_batch}', time_to_str(timer() - start_timer, 'min'), end='', flush=True)\n     \n        batch = valid_loader[t]\n        image = batch['image']\n        \n        #----\n        image = image.cuda() \n        with torch.cuda.amp.autocast(enabled=True):\n            with torch.no_grad():\n                mask = net(image)\n        #print(i, image.shape, mask.shape)\n\n        \n        \n        batch_size = len(image)\n        mask = mask.float().data.cpu().numpy()\n        for b in range(batch_size):\n            p = ((mask[b, 0]>0.4)*255).astype(np.uint8)\n\n            #---post processing ---\n            #remove small\n            #https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/456033\n            p = remove_small_objects(p, min_size=10)\n\n            #----------------------\n            rle = rle_encode(p)\n            \n            df_data.append({\n                'id': batch['id'][b],\n                'rle': rle, #'1 0', \n            })\n            \n    print('')\n    #-------------------\n    \n    df_submission = pd.DataFrame(df_data)\n    df_submission.to_csv('submission.csv', index=False)\n    print(df_submission)\n    \ndo_submit()","metadata":{"execution":{"iopub.status.busy":"2023-11-21T06:15:23.331869Z","iopub.execute_input":"2023-11-21T06:15:23.332168Z","iopub.status.idle":"2023-11-21T06:15:24.472106Z","shell.execute_reply.started":"2023-11-21T06:15:23.332145Z","shell.execute_reply":"2023-11-21T06:15:24.471097Z"},"trusted":true},"execution_count":null,"outputs":[]}]}