{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Transforming an image segmentation dataset into a classifcation dataset","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport json\nfrom pathlib import Path\nimport PIL\nimport skimage\nfrom shapely.geometry import LinearRing as ShapelyContour\nfrom shapely.geometry import Polygon as ShapelyPolygon\nimport multiprocessing\nfrom multiprocessing import Pool\nimport itertools","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\nTEST_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\nANNOT_PATH = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\nWSI_TILE_CSV = \"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\"\nWORKING_DIR = \"/kaggle/working\"\nIMG_SIZE = 512","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cleaning polygon annotations","metadata":{}},{"cell_type":"code","source":"class_names = [\"background\", \"blood_vessel\", \"glomerulus\", \"unsure\"]\n\nwsi_df = pd.read_csv(WSI_TILE_CSV)\n# Tiles with Annotations\ndataset12_ids = wsi_df[wsi_df[\"dataset\"] != 3].id.values.tolist() \n# Tiles with Expert Reviewed Annotations\ndataset1_ids = wsi_df[wsi_df[\"dataset\"] == 1].id.values.tolist()\n# Tiles with Sparse Annotations\ndataset2_ids = wsi_df[wsi_df[\"dataset\"] == 2].id.values.tolist()\n# Tiles without Annotations\ndataset3_ids = wsi_df[wsi_df[\"dataset\"] == 3].id.values.tolist() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_annotations(annotations_path=ANNOT_PATH):\n    \n    annotations = {}\n    duplicates_ids = []\n\n    with open(annotations_path, \"r\") as f:\n        for line in f:\n            entry = json.loads(line)\n            identifier = entry[\"id\"]\n            annotations_list = entry[\"annotations\"]\n\n            revised_annotations = [] # [[label_str, polygon_list], ...]\n\n            for annotation in annotations_list:\n                label = annotation[\"type\"]\n                polygon = annotation[\"coordinates\"][0]\n\n                # Remove Duplicates\n                if not any(ShapelyContour(p).equals(ShapelyContour(polygon)) and l==label for l, p in revised_annotations):\n                    revised_annotations.append([label, polygon])\n\n            if len(annotations_list) != len(revised_annotations):\n                duplicates_ids.append(identifier)\n\n            annotations[identifier] = revised_annotations\n    return annotations, duplicates_ids\n\ndef clean_annotations(annotations):\n    \n    new_annotations = {}\n    anomalies_ids = []\n\n    for identifier in annotations.keys():\n\n        annotations_list = annotations[identifier]\n        revised_annotations = []\n        skip = False\n\n        for (label, polygon) in annotations_list:\n\n            for i, (l, p) in enumerate(revised_annotations):\n                # New Polygon Inside Old Polygon -> Skip New\n                if ShapelyPolygon(p).buffer(0).contains(ShapelyPolygon(polygon).buffer(0)):\n                    skip = True\n                    break\n                # Old Polygon Inside New Polygon -> Delete Old\n                if ShapelyPolygon(p).buffer(0).within(ShapelyPolygon(polygon).buffer(0)):\n                    revised_annotations.pop(i)\n            if skip:\n                continue\n            revised_annotations.append([label, polygon])\n\n        if len(annotations_list) != len(revised_annotations):\n            anomalies_ids.append(identifier)\n\n        new_annotations[identifier] = revised_annotations\n\n    return new_annotations, anomalies_ids ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations, _ = load_annotations()\nannotations, _ = clean_annotations(annotations)\nassert len(annotations) == len(dataset12_ids)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dividing images into smaller patches","metadata":{}},{"cell_type":"code","source":"def image2patches(identifier, args):\n    \n    annotations, class_names, pos_lbl, patch_size, pos_dir, neg_dir = args\n    \n    # print(f\"Dividing {identifier}.tif into patches...\")\n    img_path = Path(TRAIN_DIR) / f\"{identifier}.tif\"\n    img = PIL.Image.open(img_path).convert(\"RGB\")\n    img = np.array(img)\n\n    mask_shape = img.shape[:2]\n    binary_mask = np.zeros(mask_shape, dtype=\"uint8\")\n    for (label, polygon) in annotations[identifier]:\n        if label == pos_lbl or pos_lbl is None:\n            label_int = class_names.index(label)\n            mask = skimage.draw.polygon2mask(mask_shape[:2], polygon).T\n            binary_mask += mask.astype(\"uint8\")\n    binary_mask = binary_mask > 0\n\n    counts = [0, 0]\n    nx, ny = 2*img.shape[0]//patch_size-1, 2*img.shape[1]//patch_size-1\n    for i in range(nx):\n        for j in range(ny):\n            x_start, y_start = i * patch_size // 2, j * patch_size // 2\n            x_end, y_end = x_start + patch_size, y_start + patch_size\n            patch = img[x_start:x_end, y_start:y_end]\n            patch_mask = binary_mask[x_start:x_end, y_start:y_end]\n            ppatch = PIL.Image.fromarray(patch)\n            if patch_mask.mean() > 0.8:\n                counts[1] += 1\n                ppatch.save(pos_dir / f\"{identifier}-{counts[1]:09d}.png\")\n            elif patch_mask.mean() < 0.1:\n                counts[0] += 1\n                ppatch.save(neg_dir / f\"{identifier}-{counts[0]:09d}.png\")\n                \n    return counts","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_gt_classification(identifiers, annotations, class_names, positive_label=None, patch_size=128, sv_dir=WORKING_DIR):\n    \"\"\"\n    Transform images into smaller patches given a list of polygon annotations \n    Patches could be used for a binary classifcation task\n    \"\"\"   \n    sv_dir = Path(sv_dir)\n    pos_dir = sv_dir / positive_label if positive_label is not None else sv_dir / \"foreground\"\n    neg_dir = sv_dir / (\"not_\" + positive_label) if positive_label is not None else sv_dir / \"background\"\n    pos_dir.mkdir(exist_ok=True, parents=True)\n    neg_dir.mkdir(exist_ok=True, parents=True)\n    \n    num_workers = multiprocessing.cpu_count()\n    print(f\"Processing using {num_workers} workers !\")  \n    \n    p = Pool(num_workers)\n    args = (annotations, class_names, positive_label, patch_size, pos_dir, neg_dir)\n    counts = p.starmap(image2patches, zip(identifiers, itertools.repeat(args, len(identifiers))))\n    counts = np.sum(counts, axis=0).tolist()\n    print(f\"Created {counts[1]} postive patches and {counts[0]} negative patches\")\n    \n    return counts","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prepare_gt_classification(dataset12_ids, annotations, class_names, positive_label=\"glomerulus\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}