{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nimport math\nimport random\nimport os\nimport csv\nfrom glob import glob\nfrom PIL import Image # For loading tiff files\nimport matplotlib.pyplot as plt\nfrom matplotlib.image import imread \n\nfrom itertools import chain\nimport json\nimport shutil\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore\nimport yaml","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-09T04:06:50.259214Z","iopub.execute_input":"2023-11-09T04:06:50.263042Z","iopub.status.idle":"2023-11-09T04:06:51.032841Z","shell.execute_reply.started":"2023-11-09T04:06:50.262982Z","shell.execute_reply":"2023-11-09T04:06:51.031750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Paths:\n    MAIN_DIR = \"/kaggle/input/blood-vessel-segmentation\"\n    TRAIN_RLES = os.path.join(MAIN_DIR, \"train_rles.csv\")\n    TRAIN = os.path.join(MAIN_DIR, \"train\")\n    TEST = os.path.join(MAIN_DIR, \"test\")","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:06:51.034572Z","iopub.execute_input":"2023-11-09T04:06:51.035222Z","iopub.status.idle":"2023-11-09T04:06:51.041474Z","shell.execute_reply.started":"2023-11-09T04:06:51.035191Z","shell.execute_reply":"2023-11-09T04:06:51.040498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=123):\n    random.seed(seed)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:06:51.042676Z","iopub.execute_input":"2023-11-09T04:06:51.043629Z","iopub.status.idle":"2023-11-09T04:06:51.054479Z","shell.execute_reply.started":"2023-11-09T04:06:51.043598Z","shell.execute_reply":"2023-11-09T04:06:51.053499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def list_files_in_directory(directory_path = Paths.TRAIN):\n    direc = []\n    for root, dirs, files in os.walk(directory_path):\n        for dire in dirs:\n            if dire in [\"labels\",\"images\"]:\n                continue \n            file_path = os.path.join(root, dire)\n            direc.append(file_path)\n            print(file_path)\n    return direc\n\nprint(\"Train Files\".center(60, \"-\"))\ntrain_folders = list_files_in_directory()\nprint(\"\\n\")\nprint(\"Test Files\".center(60, \"-\"))\ntest_folders = list_files_in_directory(directory_path=Paths.TEST)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:06:51.056805Z","iopub.execute_input":"2023-11-09T04:06:51.057623Z","iopub.status.idle":"2023-11-09T04:07:02.927162Z","shell.execute_reply.started":"2023-11-09T04:06:51.057582Z","shell.execute_reply":"2023-11-09T04:07:02.926325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_total_img(folders, test=False):\n    sub_f = [\"images\",\"labels\"] if not test else [\"images\"]\n    path = []\n    total_files = []\n    for dire in folders:\n        for subf in sub_f:\n            if (dire == \"/kaggle/input/blood-vessel-segmentation/train/kidney_3_dense\"):\n                continue \n                \n            \n            _dir = dire + \"/\" + subf\n            total_sample = len(os.listdir(_dir))\n            print(f\"{_dir}: {total_sample}\")\n            path.append(_dir)\n            total_files.append(total_sample)\n    obj = {\n        \"path\": path,\n        \"total_files\": total_files\n    }\n    return obj\n\nprint(\"Train File Objects\".center(60, \"-\"))\ntrain_file_dir = count_total_img(train_folders)\nprint(\"\\n\")\nprint(\"Test File Objects\".center(60, '-'))\ntest_file_dir = count_total_img(test_folders, test=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:07:02.929306Z","iopub.execute_input":"2023-11-09T04:07:02.929727Z","iopub.status.idle":"2023-11-09T04:07:02.947150Z","shell.execute_reply.started":"2023-11-09T04:07:02.929687Z","shell.execute_reply":"2023-11-09T04:07:02.946149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Credit to: mersico at https://www.kaggle.com/code/mersico/medical-instance-segmentation-with-yolov8\n\nIn this notebook I managed to configure the COCODataset class by mersico from a previious competition for the *SenNet + HOA -Hacking the human vasculature in 3D* competition\n\nThis is still a work and progress and more work should be done to format the data into train and val folders for the competition but it is a start. \n","metadata":{}},{"cell_type":"code","source":"class COCODataset:\n    def __init__(self, images_dirpaths: list, labels_dirpaths: list, encodings_filepath: str, length: int):\n        self.train_size = None\n        self.val_size = None\n        self.length = length\n        self.classes = None\n        self.labels_counter = None\n        self.normalize = None\n        \n        self.images_dirpaths = images_dirpaths\n        self.labels_dirpaths = labels_dirpaths\n        self.encodings_filepath = encodings_filepath\n        self.dataset_dirpath = os.path.join(os.getcwd(), \"dataset\")\n        \n        self.train_dirpath = os.path.join(self.dataset_dirpath, \"train\")\n        self.val_dirpath = os.path.join(self.dataset_dirpath, \"val\")\n        self.config_path = os.path.join(self.dataset_dirpath, \"coco.yaml\")\n        \n        self.samples = self.parse_csv(self.encodings_filepath)\n        self.image_paths_dict = self.__create_paths_dict(self.images_dirpaths)\n        self.label_paths_dict = self.__create_paths_dict(self.labels_dirpaths)\n    \n    def __prepare_dirs(self) -> None:\n        if not os.path.exists(self.dataset_dirpath):\n            os.makedirs(os.path.join(self.train_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.train_dirpath, \"labels\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"labels\"), exist_ok=True)\n            \n        else:\n            raise RuntimeError(\"Dataset already exists!\")\n    \n    def __create_paths_dict(self, dirpath) -> dict:\n        keys = [path.split(\"/\")[-2] for path in dirpath]\n        return { key: path for key, path in zip(keys, dirpath) }\n            \n            \n    def __define_splitratio(self) -> None:\n        self.train_size = round(self.length * self.train_size)\n        self.val_size = self.length - self.train_size\n        \n        assert self.train_size + self.val_size == self.length\n        \n    def parse_csv(self, path: str) -> list[dict]:\n        csv_samples = []\n\n        with open(path, 'r') as csv_file:\n            csv_reader = csv.DictReader(csv_file)\n            for row in tqdm(csv_reader, desc=\"Processing polygons\", total=self.length):\n                csv_samples.append(dict(row))\n\n        return csv_samples[:self.length]\n            \n    def __define_paths(self, i: int) -> dict:\n        data_path = self.val_dirpath\n        if i < self.train_size:\n            data_path = self.train_dirpath\n        return {\n            \"images\": os.path.join(data_path, \"images\"),\n            \"labels\": os.path.join(data_path, \"labels\")\n        }\n    \n    # Method for copying the tiff file labels to the working directory (still too large for the whole dataset)\n    # @staticmethod\n    # def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        # return os.path.join(\n            # paths_dict[\"labels\"],\n            # f\"{identifier}.tif\"\n        # )\n    \n    @staticmethod\n    def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"labels\"],\n            f\"{identifier}.txt\"\n        )\n    \n    @staticmethod\n    def __get_image_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"images\"],\n            f\"{identifier}.tif\"\n        )\n    \n    def __copy_image(self, dst_path: str, identifier: str) -> None:\n        id_split = identifier.split(\"_\")\n        folder, identifier = \"_\".join(id_split[:-1]), id_split[-1]\n        image_dirpath = self.image_paths_dict[folder]\n        shutil.copyfile(\n            os.path.join(image_dirpath, f\"{identifier}.tif\"), dst_path\n        )\n        \n        \n        # Used if copying the tif label files\n#     def __copy_label(self, dst_path: str, identifier: str) -> None:\n#         id_split = identifier.split(\"_\")\n#         folder, identifier = \"_\".join(id_split[:-1]), id_split[-1]\n#         label_dirpath = self.label_paths_dict[folder]\n#         shutil.copyfile(\n#             os.path.join(label_dirpath, f\"{identifier}.tif\"), dst_path\n#         )\n#         self.labels_counter += 1\n        \n    def __copy_label(self, encoding: list, dst_path: str) -> None:\n        with open(dst_path, \"w\") as file:\n            file.write(f\"{encoding}\\n\")\n            self.labels_counter += 1\n        \n    def __splitfolders(self):\n        for i, line in tqdm(enumerate(self.samples), desc=\"Dataset creation\", total=self.length):\n            self.labels_counter = 0\n            identifier = line[\"id\"]\n            encoding = line[\"rle\"]\n            \n            # Decides whether image is train or val\n            paths_dict = self.__define_paths(i)\n            \n            dst_image_path = self.__get_image_path(paths_dict, identifier)\n            dst_label_path = self.__get_label_path(paths_dict, identifier)\n            \n            # Image is copied from dataset to working directory\n            self.__copy_image(dst_image_path, identifier)\n            self.__copy_label(encoding, dst_label_path)\n            \n            if self.labels_counter == 0:\n                os.remove(dst_image_path)\n                os.remove(dst_label_path)\n                \n    def __count_dataset(self) -> dict:\n        train_images = len(os.listdir(os.path.join(self.train_dirpath, \"images\")))\n        train_labels = len(os.listdir(os.path.join(self.train_dirpath, \"labels\")))\n        val_images = len(os.listdir(os.path.join(self.val_dirpath, \"images\")))\n        val_labels = len(os.listdir(os.path.join(self.val_dirpath, \"labels\")))\n        \n        return {\n            \"train_images\": train_images, \n            \"train_labels\": train_labels,\n            \"val_images\": val_images,\n            \"val_labels\": val_labels\n        }\n    \n    @staticmethod\n    def __check_sanity(count_dict: dict) -> None:\n        assert count_dict[\"train_images\"] == count_dict[\"train_labels\"]\n        assert count_dict[\"val_images\"] == count_dict[\"val_labels\"]\n    \n    def __finalizing(self, count_dict: dict) -> None:\n        assert os.path.exists(self.dataset_dirpath)\n        \n        example_structure = [\n            \"dataset\",\n            \"train\", \"labels\", \"images\",\n            \"val\", \"labels\", \"images\"\n        ]\n        \n        dir_bone = (\n            dirname.split(\"/\")[-1]\n            for dirname, _, filenames in os.walk(self.dataset_dirpath)\n            if dirname.split(\"/\")[-1] in example_structure\n        )\n        \n        try:\n            print(\"\\n~ HuBMAP Dataset Structure ~\\n\")\n            print(\n            f\"\"\"\n          ├── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n            \"\"\"\n            )\n        except StopIteration as e:\n            print(e)\n        else:\n            print(Fore.GREEN + \"-> Success\")\n            print(Fore.GREEN + f\"Train dataset: {count_dict['train_images']}\\nVal dataset: {count_dict['val_images']}\")\n    \n    def get_config(self) -> dict:\n        return {\n            \"train\": str(self.train_dirpath),\n            \"val\": str(self.val_dirpath)\n        }\n    \n    @staticmethod\n    def display_config(config: dict) -> None:\n        print(Fore.BLACK + \"\\n~ HuBMAP Config Structure ~\\n\")\n        print(\n        f\"\"\"\n      │   │\n      │   ├── train\n      │   │   └── {config['train']}/images\n      │   │\n      │   │\n      │   ├── val\n      │   │   └── {config['val']}/images\n        \"\"\"\n        )\n        print(Fore.GREEN + \"-> Success\")\n    \n    def write_config(self, config: dict) -> None:\n        with open(self.config_path, mode=\"w\") as f:\n            yaml.safe_dump(stream=f, data=config)\n            \n    def __call__(self, train_size: float, make_config: bool = True, \n                normalize: bool = True) -> None:\n        self.train_size = train_size\n        self.normalize = normalize\n        \n        self.__define_splitratio()\n        self.__prepare_dirs()\n        self.__splitfolders()\n        count_dict = self.__count_dataset()\n        self.__check_sanity(count_dict)\n        self.__finalizing(count_dict)\n        \n        if make_config: \n            config = self.get_config()\n            self.write_config(config)\n            self.display_config(config)\n                     ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:07:02.948708Z","iopub.execute_input":"2023-11-09T04:07:02.949030Z","iopub.status.idle":"2023-11-09T04:07:02.981349Z","shell.execute_reply.started":"2023-11-09T04:07:02.949003Z","shell.execute_reply":"2023-11-09T04:07:02.980206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rles = pd.read_csv(Paths.TRAIN_RLES)\ndataset_len = len(train_rles)\ntrain_image_paths = train_file_dir[\"path\"][::2]\ntrain_label_paths = train_file_dir[\"path\"][1::2]","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:07:02.983563Z","iopub.execute_input":"2023-11-09T04:07:02.984060Z","iopub.status.idle":"2023-11-09T04:07:04.168760Z","shell.execute_reply.started":"2023-11-09T04:07:02.984033Z","shell.execute_reply":"2023-11-09T04:07:04.167695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco = COCODataset(train_image_paths, train_label_paths, Paths.TRAIN_RLES, length=2000)\ncoco(0.8, make_config=True, normalize=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:07:04.170234Z","iopub.execute_input":"2023-11-09T04:07:04.170563Z","iopub.status.idle":"2023-11-09T04:08:38.640776Z","shell.execute_reply.started":"2023-11-09T04:07:04.170535Z","shell.execute_reply":"2023-11-09T04:08:38.639816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}