{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Contents\n[1. Info](#1.-Info)\n - 1-1. Pre-processed data\n - 1-2. Trained Models\n - 1-3. Datasets\n    \n    \n[2. Used Packages](#2.-Used-Packages) \n - 2-1. Install & Import Packages\n - 2-2. Define Functions\n    \n    \n[3. Models](#3.-Models)\n - Set 2 kind of Fold Model Groups\n - 3-1. fold_models_1 : 'hubmap-tf-with-tpu-efficientunet-512x512-train'\n - 3-2. fold_models_2 : 'hubmap-models-cv-08848-pl-0847' \n\n\n4. Inference (fold model predict)\n\n\n- Save Submission File","metadata":{}},{"cell_type":"markdown","source":"-----\n# 1. Info\n* 1-1. Pre-processed data : This Notebook is summarized with other trials\n* 1-2. Image Data Pre-Processing & Basic Model, based on [Wojtek's work](https://www.kaggle.com/wrrosa/hubmap-tf-with-tpu-efficientunet-512x512-train) \n    * Model Prediction, based on [Ashish Gupta's work](https://www.kaggle.com/roydatascience/hubmap-sub-effunet5-tpu-efficientunet-512x512)\n* 1-3. Before you start, make sure Datasets are added on your Notebook\n    * for package installation : 'kerasapplications' , 'efficientnet'\n    * for trained models       : 'hubmap-models-cv-0.8848-pl=0.847' , 'hubmap-tf-with-tpu-efficientunet-512x512' \n* (Optional) if you want to know how to control tiff large images, check this link [Nihad TP's work](https://www.kaggle.com/nihadtp/kidney-hacking-exploration-stage)","metadata":{}},{"cell_type":"markdown","source":"-----\n# 2. Used Packages\n\n* 2-1. Install & Import Packages\n* 2-2. Define Functions","metadata":{}},{"cell_type":"markdown","source":"> ---\n> ### 2-1. Install & Import Packages","metadata":{}},{"cell_type":"code","source":"# Install 'Keras Applicatoins' & 'Efficientnet' Packages\n! pip install ../input/kerasapplications/keras-team-keras-applications-3b180cb -f ./ --no-index -q\n! pip install ../input/efficientnet/efficientnet-1.1.0/ -f ./ --no-index -q\n\n# Import Installed Packages\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport gc\n\nimport rasterio\nfrom rasterio.windows import Window\n\nimport pathlib\nfrom tqdm.notebook import tqdm\nimport cv2\n\nimport tensorflow as tf\nimport efficientnet as efn\nimport efficientnet.tfkeras\n\nimport os, glob, gc\nimport json\n\nosj = os.path.join # function to merge dir names","metadata":{"execution":{"iopub.status.busy":"2021-06-23T01:06:02.847200Z","iopub.execute_input":"2021-06-23T01:06:02.847506Z","iopub.status.idle":"2021-06-23T01:06:24.617509Z","shell.execute_reply.started":"2021-06-23T01:06:02.847432Z","shell.execute_reply":"2021-06-23T01:06:24.616734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ---\n> ### 2-2. Define Functions","metadata":{}},{"cell_type":"code","source":"# Functions from Wojteck Kernel\n\ndef rle_encode_less_memory(img):\n    pixels = img.T.flatten()\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef make_grid(shape, window=256, min_overlap=32):\n    \"\"\"\n        Return Array of size (N,4), where N - number of tiles,\n        2nd axis represente slices: x1,x2,y1,y2 \n    \"\"\"\n    x, y = shape\n    nx = x // (window - min_overlap) + 1\n    x1 = np.linspace(0, x, num=nx, endpoint=False, dtype=np.int64)\n    x1[-1] = x - window\n    x2 = (x1 + window).clip(0, x)\n    ny = y // (window - min_overlap) + 1\n    y1 = np.linspace(0, y, num=ny, endpoint=False, dtype=np.int64)\n    y1[-1] = y - window\n    y2 = (y1 + window).clip(0, y)\n    slices = np.zeros((nx,ny, 4), dtype=np.int64)\n    \n    for i in range(nx):\n        for j in range(ny):\n            slices[i,j] = x1[i], x2[i], y1[j], y2[j]    \n    return slices.reshape(nx*ny,4)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T01:06:32.901646Z","iopub.execute_input":"2021-06-23T01:06:32.901978Z","iopub.status.idle":"2021-06-23T01:06:32.912524Z","shell.execute_reply.started":"2021-06-23T01:06:32.901948Z","shell.execute_reply":"2021-06-23T01:06:32.911383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-----\n\n# 3. Models\n\n* 3-1. fold_models_1 : 'hubmap-tf-with-tpu-efficientunet-512x512-train'\n* 3-2. fold_models_2 : 'hubmap-models-cv-08848-pl-0847'\n","metadata":{}},{"cell_type":"markdown","source":"> ---\n> ### 3-1. fold_models_1 : 'hubmap-tf-with-tpu-efficientunet-512x512-train'","metadata":{}},{"cell_type":"code","source":"import yaml    # package for read yaml file\nimport pprint  # package for simplified prompt output print\n\n# Dir including 'Model' , 'metrics' , 'params'\nmod_path = '/kaggle/input/hubmap-tf-with-tpu-efficientunet-512x512-train/'\n\n# Load parameters\nwith open(mod_path+'params.yaml') as file:\n    P = yaml.load(file, Loader=yaml.FullLoader)\n    pprint.pprint(P)\n\n# Define additional parameters    \nTHRESHOLD = 0.4\nWINDOW = 1024\nMIN_OVERLAP = 300\nNEW_SIZE = P['DIM']","metadata":{"execution":{"iopub.status.busy":"2021-06-23T01:06:34.352209Z","iopub.execute_input":"2021-06-23T01:06:34.352539Z","iopub.status.idle":"2021-06-23T01:06:34.377267Z","shell.execute_reply.started":"2021-06-23T01:06:34.352513Z","shell.execute_reply":"2021-06-23T01:06:34.376494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport json\n\n# Load METRICS, information of Models \nwith open(mod_path + 'metrics.json') as json_file:\n    M = json.load(json_file)\n    json_file.close()\n    \nM_info = pd.DataFrame(M)\nM_info = M_info[['val_loss','val_dice_coe','val_accuracy']]\n\nprint('[ Info of Models ]')\nprint('%-18s : %s' % ('Model Updated Date',str(M['datetime'])))\nprint('%-18s : %s' % ('Average accuracy',str(M['oof_dice_coe'])))\nM_info.head()\n# print('%20s : ')\n\n# print('Model run datetime: '+M['datetime'])\n# print('OOF val_dice_coe: ' + str(M['oof_dice_coe']))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T01:06:39.368999Z","iopub.execute_input":"2021-06-23T01:06:39.369331Z","iopub.status.idle":"2021-06-23T01:06:39.437049Z","shell.execute_reply.started":"2021-06-23T01:06:39.369300Z","shell.execute_reply":"2021-06-23T01:06:39.436197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define coordinates\n# | x' |   | a b c | | x |\n# | y' | = | d e f | | y |\n# | 1  |   | 0 0 1 | | 1 |\nidentity = rasterio.Affine(1, 0, 0, 0, 1, 0)\n\n# make model list for cross validate the models\nfold_models_1 = []\n# Dir including 'Model' , 'metrics' , 'params'\nmod_path = '/kaggle/input/hubmap-tf-with-tpu-efficientunet-512x512-train/'\n\nfor fold_model_path in glob.glob(mod_path+'*.h5'):\n    fold_models_1.append(tf.keras.models.load_model(fold_model_path,compile = False))\nprint('Target Fold Models Group 1 : %d EA' % len(fold_models_1))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T01:06:40.350071Z","iopub.execute_input":"2021-06-23T01:06:40.350427Z","iopub.status.idle":"2021-06-23T01:07:10.760006Z","shell.execute_reply.started":"2021-06-23T01:06:40.350398Z","shell.execute_reply":"2021-06-23T01:07:10.759047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ---\n> ### 3-2. fold_models_2 : 'hubmap-models-cv-08848-pl-0847'","metadata":{}},{"cell_type":"code","source":"# Parameters from ISA's Kernel\n\ndebug = True # True False\nn_debug_images = 1 if debug else 1000000000\nn_debug_slices = 20 if debug else 1000000000\n\n# whether to run prediction when committing. WILL RUN predictions during submission in any case\ndo_predict = False if not debug else True\n\nmodels_dir = '../input/hubmap-models-cv-08848-pl-0847'\nmodel_filepaths = [ os.path.join(models_dir, f\"model-fold-{i}.h5\") for i in range(4)]\n\nassert len(model_filepaths)==len(np.unique(model_filepaths))\n#folds_to_predict = [i for (i, fn) in enumerate(model_filepaths) if os.path.isfile(fn)]\nmodel_dirnames = [os.path.dirname(filepath) for filepath in model_filepaths]\n\n#check_order = [fn.split('.')[-2].split('-')[-1] == i for (i,fn) in enumerate(model_filepaths) if fn.strip()!='']\n#assert np.sum(check_order)==0, 'models should be in folds order or empty string'\n\nimport yaml\nimport pprint\nwith open(osj(model_dirnames[0],'params.yaml')) as file:\n    P = yaml.load(file, Loader=yaml.FullLoader)\n    pprint.pprint(P)\n\nTHRESHOLD = 0.30\nWINDOW = 1024\nMIN_OVERLAP = 32 \nNEW_SIZE = P['DIM']\n\nassert sum([not os.path.isfile(path_) for path_ in model_filepaths]) == 0\nprint(\"\\n Number of models : {}\".format(len(model_filepaths)))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T02:08:28.386950Z","iopub.execute_input":"2021-06-23T02:08:28.387296Z","iopub.status.idle":"2021-06-23T02:08:28.412312Z","shell.execute_reply.started":"2021-06-23T02:08:28.387265Z","shell.execute_reply":"2021-06-23T02:08:28.411203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# About the Models\n\nave_score = 0\nfor i, m_path in enumerate(model_filepaths):\n    fold_ = int(m_path.split('.')[-2].split('-')[-1])\n    with open(osj(model_dirnames[i],'metrics.json')) as json_file:\n        M = json.load(json_file)\n    print(f\"\\n ----------- \\nModel {model_dirnames[i].split('/')[-1]}\" +\n          '\\nval_dice_coe: '+ str(round(M['val_dice_coe'][fold_], 5)) +\n          '\\tval_loss: ' + str(round(M['val_loss'][fold_], 5)) +\n          '\\tval_accuracy: '+ str(round(M['val_accuracy'][fold_], 5))\n          )\n\n\n# See the Model Informations\nfor model_group in np.unique(model_dirnames):\n    with open(osj(model_group,'metrics.json')) as json_file:\n        M = json.load(json_file)\n        ave_dice = np.mean(M['val_dice_coe']) \n    ave_loss = np.mean(M['val_loss'])  # /len(folds_to_predict)\n    ave_accuracy = np.mean(M['val_accuracy'])\n    print(f\"\\n ============ MODEL GROUP {model_group} ==============\")\n    print(\" ------------ \\nAVERAGE DICE SCORE = {}\".format(round(ave_dice, 5)))\n    print(\" ------------ \\nAVERAGE VALIDATION LOSS = {}\".format(round(ave_loss, 5)))\n    print(\" ------------ \\nAVERAGE VALIDATION ACCURACY = {}\".format(round(ave_accuracy, 5)))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T02:02:53.649859Z","iopub.execute_input":"2021-06-23T02:02:53.650199Z","iopub.status.idle":"2021-06-23T02:02:53.662917Z","shell.execute_reply.started":"2021-06-23T02:02:53.650161Z","shell.execute_reply":"2021-06-23T02:02:53.662004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Make fold model list \n\nif do_predict:\n    identity = rasterio.Affine(1, 0, 0, 0, 1, 0)\n    fold_models_2 = []\n    \n    for fold_model_path in model_filepaths:\n        fold_models_2.append(tf.keras.models.load_model(fold_model_path,compile = False))\n#     print(len(fold_models_2))\n#     print('Target Fold Models Group 2 : %d EA' % len(model_filepaths))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T02:02:53.825286Z","iopub.execute_input":"2021-06-23T02:02:53.825552Z","iopub.status.idle":"2021-06-23T02:03:21.414820Z","shell.execute_reply.started":"2021-06-23T02:02:53.825528Z","shell.execute_reply":"2021-06-23T02:03:21.413859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-----\n# 4. Inference (fold model predict)","metadata":{}},{"cell_type":"code","source":"print('Target Fold Model Group 1 : %d EA' % len(fold_models_1))\nprint('Target Fold Model Group 2 : %d EA' % len(fold_models_2))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T02:03:21.416973Z","iopub.execute_input":"2021-06-23T02:03:21.417542Z","iopub.status.idle":"2021-06-23T02:03:21.423089Z","shell.execute_reply.started":"2021-06-23T02:03:21.417501Z","shell.execute_reply":"2021-06-23T02:03:21.422199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testset path\np = pathlib.Path('../input/hubmap-kidney-segmentation')\nsubm = {}\n\n\n###########################################\n###### If you intend to prediction ########\nskip_pred = False ## change it to \"False\"##\n###########################################\nif skip_pred!=True:\n    for i, filename in tqdm(enumerate(p.glob('test/*.tiff')), \n                            total = len(list(p.glob('test/*.tiff')))):\n\n        print(f'{i+1} Predicting {filename.stem}')\n\n        dataset = rasterio.open(filename.as_posix(), transform = identity)\n#         print(dataset.shape)\n        \n        slices = make_grid(dataset.shape, WINDOW, min_overlap=MIN_OVERLAP)\n#         print(slices.shape)\n        preds = np.zeros(dataset.shape, dtype=np.uint8)\n\n        for (x1,x2,y1,y2) in slices:\n            image = dataset.read(1, # [1,2,3]\n                                 window=Window.from_slices((x1,x2),(y1,y2),boundless=True)) \n            image = np.moveaxis(image, 0, -1)\n            image = cv2.resize(image, (NEW_SIZE, NEW_SIZE),interpolation = cv2.INTER_AREA)\n            image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)\n            image = np.expand_dims(image, 0)\n\n            pred_1 = None\n            pred_2 = None\n\n            # Predict with fold_models_1 models\n\n            for fold_model in fold_models_1:\n                if pred_1 is None:\n                    pred_1 = np.squeeze(fold_model.predict(image))\n                else:\n                    pred_1 += np.squeeze(fold_model.predict(image))\n\n            # Predict with fold_models_2 models\n\n            for fold_model in fold_models_2:\n                if pred_2 is None:\n                    pred_2 = np.squeeze(fold_model.predict(image))\n                else:\n                    pred_2 += np.squeeze(fold_model.predict(image))\n\n            pred_1 = pred_1/len(fold_models_1)\n            pred_2 = pred_2/len(fold_models_2)\n\n            # calculate average of pred results\n            pred = 0.5 * pred_1 + 0.5 * pred_2\n\n            pred = cv2.resize(pred, (WINDOW, WINDOW))\n            preds[x1:x2,y1:y2] += (pred > THRESHOLD).astype(np.uint8)\n\n        preds = (preds > 0.5).astype(np.uint8)\n\n        subm[i] = {'id':filename.stem, 'predicted': rle_encode_less_memory(preds)}\n        print(np.sum(preds))\n        del preds\n        gc.collect();","metadata":{"execution":{"iopub.status.busy":"2021-06-23T02:08:49.755148Z","iopub.execute_input":"2021-06-23T02:08:49.755483Z","iopub.status.idle":"2021-06-23T03:17:00.942798Z","shell.execute_reply.started":"2021-06-23T02:08:49.755454Z","shell.execute_reply":"2021-06-23T03:17:00.942104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(slices.shape)\nprint(dataset.shape)\ntmp_1 = dataset.read(1,window=Window.from_slices((x1,x2),(y1,y2)))\ntmp_2 = dataset.read(2,window=Window.from_slices((x1,x2),(y1,y2)))\ntmp_3 = dataset.read(3,window=Window.from_slices((x1,x2),(y1,y2)))\nfig, ax = plt.subplots(1,3,figsize=(10,5))\nax[0].imshow(tmp_1)\nax[1].imshow(tmp_2)\nax[2].imshow(tmp_3)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:20:08.833106Z","iopub.execute_input":"2021-06-23T03:20:08.833486Z","iopub.status.idle":"2021-06-23T03:20:08.873761Z","shell.execute_reply.started":"2021-06-23T03:20:08.833457Z","shell.execute_reply":"2021-06-23T03:20:08.872239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-----\n# Save Submission File","metadata":{}},{"cell_type":"code","source":"# see submission samples\n\nimport pandas as pd\nsample_sub = pd.read_csv('../input/hubmap-kidney-segmentation/sample_submission.csv')\nprint(sample_sub.shape)\nsample_sub","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:20:42.610327Z","iopub.execute_input":"2021-06-23T03:20:42.610656Z","iopub.status.idle":"2021-06-23T03:20:42.630201Z","shell.execute_reply.started":"2021-06-23T03:20:42.610627Z","shell.execute_reply":"2021-06-23T03:20:42.629174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_info = pd.read_csv('../input/hubmap-kidney-segmentation/HuBMAP-20-dataset_information.csv')\ndata_info","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:17:00.944676Z","iopub.execute_input":"2021-06-23T03:17:00.945031Z","iopub.status.idle":"2021-06-23T03:17:00.985770Z","shell.execute_reply.started":"2021-06-23T03:17:00.944994Z","shell.execute_reply":"2021-06-23T03:17:00.985090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# run after prediction\n\nsubmission = pd.DataFrame.from_dict(subm, orient='index')\nsubmission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2021-06-23T03:20:53.804231Z","iopub.execute_input":"2021-06-23T03:20:53.804571Z","iopub.status.idle":"2021-06-23T03:20:54.133819Z","shell.execute_reply.started":"2021-06-23T03:20:53.804540Z","shell.execute_reply":"2021-06-23T03:20:54.133120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}