{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### TensorFlow UNet submittiom\n\n#### Related dataset & notebook:\n* https://www.kaggle.com/joshi98kishan/hubmap-keras-pipeline-training-inference\n* https://www.kaggle.com/leighplt/pytorch-fcn-resnet50\n\nUpdate (19.08.2022): efficientnetb6 and 512 blocks","metadata":{}},{"cell_type":"markdown","source":"#### Read parameteres from notebook output, actually only DIM is used:","metadata":{}},{"cell_type":"code","source":"mod_path = '../input/tpuhubmaphpa-train-tf-full3c/'\nimport yaml\nimport pprint\nwith open(mod_path+'params.yaml') as file:\n    P = yaml.load(file, Loader=yaml.FullLoader)\n    pprint.pprint(P)\nWINDOW = 2048\nMIN_OVERLAP = 1024\nNEW_SIZE = P['DIM']","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:49:08.202429Z","iopub.execute_input":"2022-08-19T09:49:08.202812Z","iopub.status.idle":"2022-08-19T09:49:08.276763Z","shell.execute_reply.started":"2022-08-19T09:49:08.202725Z","shell.execute_reply":"2022-08-19T09:49:08.275748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Packages","metadata":{}},{"cell_type":"markdown","source":"Copy to work dir","metadata":{}},{"cell_type":"code","source":"! cp -r ../input/kerasapplications /kaggle/working\n! cp -r ../input/efficientnet /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:49:08.278649Z","iopub.execute_input":"2022-08-19T09:49:08.279121Z","iopub.status.idle":"2022-08-19T09:49:10.794026Z","shell.execute_reply.started":"2022-08-19T09:49:08.279081Z","shell.execute_reply":"2022-08-19T09:49:10.792513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/working/kerasapplications/keras-team-keras-applications-3b180cb -f ./ --no-index -q\n!pip install /kaggle/working/efficientnet/efficientnet-1.1.0/ -f ./ --no-index -q\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport gc\nimport json\n\nimport numba\n\nimport rasterio\nfrom rasterio.windows import Window\n\nimport pathlib\nfrom tqdm.notebook import tqdm\nimport cv2\n\nimport tensorflow as tf\nimport efficientnet as efn\nimport efficientnet.tfkeras","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:49:10.800028Z","iopub.execute_input":"2022-08-19T09:49:10.800361Z","iopub.status.idle":"2022-08-19T09:49:41.517273Z","shell.execute_reply.started":"2022-08-19T09:49:10.800331Z","shell.execute_reply":"2022-08-19T09:49:41.516214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Functions","metadata":{}},{"cell_type":"code","source":"def make_grid(shape, window=256, min_overlap=32):\n    \"\"\"\n        Return Array of size (N,4), where N - number of tiles,\n        2nd axis represente slices: x1,x2,y1,y2 \n    \"\"\"\n    x, y = shape\n    nx = x // (window - min_overlap)\n    x1 = np.linspace(0, x, num=nx, endpoint=False, dtype=np.int64)\n    x1[-1] = x - window\n    x2 = (x1 + window).clip(0, x)\n    ny = y // (window - min_overlap) \n    y1 = np.linspace(0, y, num=ny, endpoint=False, dtype=np.int64)\n    y1[-1] = y - window\n    y2 = (y1 + window).clip(0, y)\n    slices = np.zeros((nx,ny, 4), dtype=np.int64)\n    \n    for i in range(nx):\n        for j in range(ny):\n            slices[i,j] = x1[i], x2[i], y1[j], y2[j]    \n    return slices.reshape(nx*ny,4)\n#https://www.kaggle.com/bguberfain/memory-aware-rle-encoding\n#with transposed mask\ndef rle_encode_less_memory(img):\n    #the image should be transposed\n    pixels = img.T.flatten()\n    \n    # This simplified method requires first and last pixel to be zero\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:49:41.519955Z","iopub.execute_input":"2022-08-19T09:49:41.520890Z","iopub.status.idle":"2022-08-19T09:49:41.533312Z","shell.execute_reply.started":"2022-08-19T09:49:41.520849Z","shell.execute_reply":"2022-08-19T09:49:41.530234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Models","metadata":{}},{"cell_type":"code","source":"identity = rasterio.Affine(1, 0, 0, 0, 1, 0)\nfold_models = []\ni = 0\nfor fold_model_path in glob.glob(mod_path+'*.h5'):\n    i+=1\n    if i !=2 and i != 3:\n        continue\n    fold_models.append(tf.keras.models.load_model(fold_model_path,compile = False))\nprint(len(fold_models))","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:49:41.535098Z","iopub.execute_input":"2022-08-19T09:49:41.535517Z","iopub.status.idle":"2022-08-19T09:50:00.158206Z","shell.execute_reply.started":"2022-08-19T09:49:41.535472Z","shell.execute_reply":"2022-08-19T09:50:00.156383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Results","metadata":{}},{"cell_type":"code","source":"p = pathlib.Path('../input/hubmap-organ-segmentation')\ns_th = 40  #saturation blancking threshold\np_th = 200\n\nnames,predsres = [],[]\nfor i, filename in tqdm(enumerate(p.glob('test_images/*.tiff')), \n                        total = len(list(p.glob('test_images/*.tiff')))):\n    \n    print(f'{i+1} Predicting {filename.stem}')    \n    dataset = rasterio.open(filename.as_posix(), transform = identity)\n    preds = np.zeros(dataset.shape, dtype=np.uint8)\n    if dataset.count != 3:\n        print('Image file with subdatasets as channels')\n        layers = [rasterio.open(subd) for subd in dataset.subdatasets]      \n    if dataset.shape[0] < WINDOW or dataset.shape[1] < WINDOW:\n        x1 = 0\n        x2 = dataset.shape[1]\n        y1 = 0\n        y2 = dataset.shape[0]\n        if dataset.count == 3:\n            image = dataset.read([1,2,3],\n                            window=Window.from_slices((x1,x2),(y1,y2)))\n            image = np.moveaxis(image, 0, -1)\n        else:\n            image = np.zeros((dataset.shape[0], dataset.shape[1], 3), dtype=np.uint8)\n            for fl in range(3):\n                image[:,:,fl] = layers[fl].read(window=Window.from_slices((x1,x2),(y1,y2)))           \n        image = cv2.resize(image, (NEW_SIZE, NEW_SIZE),interpolation = cv2.INTER_AREA)        \n        image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)                                    \n        image = np.expand_dims(image, 0)\n        pred = None        \n        for fold_model in fold_models:\n            if pred is None:\n                pred = np.squeeze(fold_model.predict(image))\n            else:\n                pred += np.squeeze(fold_model.predict(image))        \n        pred = pred/len(fold_models)        \n        \n        pred = cv2.resize(pred, (dataset.shape[1], dataset.shape[0]))\n        preds[x1:x2,y1:y2] =(pred > 0.5).astype(np.uint8)            \n    else:\n        slices = make_grid(dataset.shape, window=WINDOW, min_overlap=MIN_OVERLAP)            \n        for (x1,x2,y1,y2) in slices:                     \n            if dataset.count == 3:\n                image = dataset.read([1,2,3],\n                                window=Window.from_slices((x1,x2),(y1,y2)))\n                image = np.moveaxis(image, 0, -1)\n            else:\n                image = np.zeros((WINDOW, WINDOW, 3), dtype=np.uint8)\n                for fl in range(3):\n                    image[:,:,fl] = layers[fl].read(window=Window.from_slices((x1,x2),(y1,y2)))                        \n            image = cv2.resize(image, (NEW_SIZE, NEW_SIZE),interpolation = cv2.INTER_AREA)        \n            image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)                            \n            hsv = cv2.cvtColor(image, cv2.COLOR_BGR2HSV)\n            h_, s, v = cv2.split(hsv)\n            if (s>s_th).sum() <= p_th or image.sum() <= p_th:             \n                continue        \n            image = np.expand_dims(image, 0)\n\n            pred = None\n\n            for fold_model in fold_models:\n                if pred is None:\n                    pred = np.squeeze(fold_model.predict(image))\n                else:\n                    pred += np.squeeze(fold_model.predict(image))        \n            pred = pred/len(fold_models)        \n\n            pred = cv2.resize(pred, (WINDOW, WINDOW))\n            preds[x1:x2,y1:y2] |=(pred > 0.5).astype(np.uint8)        \n    names.append(filename.stem)\n    predsres.append(rle_encode_less_memory(preds))    \n    gc.collect();        \n    \n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:50:00.159820Z","iopub.execute_input":"2022-08-19T09:50:00.160179Z","iopub.status.idle":"2022-08-19T09:50:14.749162Z","shell.execute_reply.started":"2022-08-19T09:50:00.160143Z","shell.execute_reply":"2022-08-19T09:50:14.748161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'id':names,'rle':predsres})\ndf.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:50:14.750611Z","iopub.execute_input":"2022-08-19T09:50:14.751793Z","iopub.status.idle":"2022-08-19T09:50:14.764054Z","shell.execute_reply.started":"2022-08-19T09:50:14.751754Z","shell.execute_reply":"2022-08-19T09:50:14.763173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:50:14.765726Z","iopub.execute_input":"2022-08-19T09:50:14.766107Z","iopub.status.idle":"2022-08-19T09:50:14.790295Z","shell.execute_reply.started":"2022-08-19T09:50:14.766072Z","shell.execute_reply":"2022-08-19T09:50:14.789334Z"},"trusted":true},"execution_count":null,"outputs":[]}]}