{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"[As discussed from the beginning of this competition](https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/279488), some annotation masks were broken.\nCorrecting these broken masks, either manually or automatically (or simply ignore in loss calculation), may help to improve model performance.\nTo begin with, let's discover these broken masks (candidates).","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport scipy.ndimage as ndi\nimport cv2\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-11-23T15:14:32.833475Z","iopub.execute_input":"2021-11-23T15:14:32.834348Z","iopub.status.idle":"2021-11-23T15:14:33.272441Z","shell.execute_reply.started":"2021-11-23T15:14:32.834211Z","shell.execute_reply":"2021-11-23T15:14:33.271459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/coldfir3/efficient-coco-dataset-generator/notebook\ndef rle2mask(rle, img_w, img_h):\n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype=np.uint)\n    array = array.reshape((-1, 2)).T\n    array[0] = array[0] - 1\n\n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype=np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype=np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n\n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-11-23T15:14:33.274049Z","iopub.execute_input":"2021-11-23T15:14:33.274307Z","iopub.status.idle":"2021-11-23T15:14:33.282347Z","shell.execute_reply.started":"2021-11-23T15:14:33.274275Z","shell.execute_reply":"2021-11-23T15:14:33.281401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TH = 40\n\ndata_root = Path(\"../input/sartorius-cell-instance-segmentation\")\ndf = pd.read_csv(str(data_root.joinpath(\"train.csv\")))\nimg_dir = data_root.joinpath(\"train\")\n\nfor idx, row in df.iterrows():\n    if idx > 5000:\n        break\n    \n    mask = rle2mask(row['annotation'], row['width'], row['height'])\n    mask = ndi.binary_fill_holes(mask).astype(np.uint8)\n    contours, hierarchy = cv2.findContours(mask, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)\n    c = contours[0][:, 0]\n    diff = c - np.roll(c, 1, 0)\n    targets = (diff[:, 1] == 0) & (np.abs(diff[:, 0]) >= TH)  # find horizontal lines longer than threshold\n    \n    if targets.sum() == 0:\n        continue\n        \n    if np.all(c[targets][:, 1] == 0) or np.all(c[targets][:, 1] == 519):  # remove screen edge cases\n        continue\n\n    img_id = row[\"id\"]\n    img_path = img_dir.joinpath(f\"{img_id}.png\")\n    img = cv2.imread(str(img_path), 0)\n    plt.figure(figsize=(16, 12))\n    plt.title(f\"{img_id} - {idx}\")\n    plt.imshow(img)\n    plt.imshow(mask, alpha=0.5)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-23T15:14:33.283601Z","iopub.execute_input":"2021-11-23T15:14:33.283856Z","iopub.status.idle":"2021-11-23T15:16:34.550752Z","shell.execute_reply.started":"2021-11-23T15:14:33.283817Z","shell.execute_reply":"2021-11-23T15:16:34.549906Z"},"trusted":true},"execution_count":null,"outputs":[]}]}