{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52279,"databundleVersionId":5822112,"sourceType":"competition"},{"sourceId":6021726,"sourceType":"datasetVersion","datasetId":3446188}],"dockerImageVersionId":30512,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Prepare data\n","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #abffd1; font-family:verdana; color: #003819; border: 2px #003819 solid\">\n    <b>Libraries for creating dataset</b>\n</div>","metadata":{}},{"cell_type":"code","source":"from itertools import chain\nimport json\nimport os\nimport shutil\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore\nimport yaml\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:07:40.791226Z","iopub.execute_input":"2023-11-24T16:07:40.791519Z","iopub.status.idle":"2023-11-24T16:07:40.878832Z","shell.execute_reply.started":"2023-11-24T16:07:40.791493Z","shell.execute_reply":"2023-11-24T16:07:40.877919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #abffd1; font-family:verdana; color: #003819; border: 2px #003819 solid\">\n    <b>Class for creating dataset</b>\n</div>","metadata":{}},{"cell_type":"code","source":"class COCODataset:\n    def __init__(self, images_dirpath: str, annotations_filepath: str, length: int = 1633):\n        self.train_size = None\n        self.val_size = None\n        self.length = length\n        self.classes = None\n        self.labels_counter = None\n        self.normalize = None\n        \n        self.images_dirpath = images_dirpath\n        self.annotations_filepath = annotations_filepath\n        self.dataset_dirpath = os.path.join(os.getcwd(), \"dataset\")\n        self.train_dirpath =  os.path.join(self.dataset_dirpath, \"train\")\n        self.val_dirpath =  os.path.join(self.dataset_dirpath, \"val\")\n        self.config_path = os.path.join(self.dataset_dirpath, \"coco.yaml\")\n\n        self.samples = self.parse_jsonl(annotations_filepath)\n        self.classes_dict = {\n            \"blood_vessel\": 0,\n            \"glomerulus\": 1,\n            \"unsure\": 2,\n        }\n\n    def __prepare_dirs(self) -> None:\n        if not os.path.exists(self.dataset_dirpath):\n            os.makedirs(os.path.join(self.train_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.train_dirpath, \"labels\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"labels\"), exist_ok=True)\n        else:\n            raise RuntimeError(\"Dataset already exists!\")\n\n    def __define_splitratio(self) -> None:\n        self.train_size = round(self.length * self.train_size)\n        self.val_size = self.length - self.train_size\n        assert self.train_size + self.val_size == self.length\n\n    def parse_jsonl(self, path: str) -> list[dict, ...]:\n        with open(path, 'r') as json_file:\n            jsonl_samples = [\n                json.loads(line)\n                for line in tqdm(\n                    json_file, desc=\"Processing polygons\", total=self.length\n                )\n            ]\n        return jsonl_samples\n\n    def __define_paths(self, i: int) -> dict:\n        data_path = self.val_dirpath\n        if i < self.train_size:\n            data_path = self.train_dirpath\n        return {\n            \"images\": os.path.join(data_path, \"images\"),\n            \"labels\": os.path.join(data_path, \"labels\")\n        }\n\n    @staticmethod\n    def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"labels\"],\n            f\"{identifier}.txt\"\n        )\n\n    @staticmethod\n    def __get_image_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"images\"],\n            f\"{identifier}.tif\"\n        )\n\n    def __copy_image(self, dst_path: str, identifier: str) -> str:\n        shutil.copyfile(\n            os.path.join(self.images_dirpath, f\"{identifier}.tif\"),\n            dst_path\n        )\n\n    def __copy_label(self, annotations: list, dst_path: str) -> None:\n        with open(dst_path, \"w\") as file:\n            for annotation in annotations:\n                coordinates = annotation[\"coordinates\"][0]\n                label = self.classes_dict[annotation[\"type\"]]\n                if label in self.classes:\n                    if coordinates:\n                        if self.normalize:\n                            coordinates = np.array(coordinates) / 512.0\n                        coordinates = \" \".join(map(str, chain(*coordinates)))\n                        file.write(f\"{label} {coordinates}\\n\")\n                        self.labels_counter += 1\n\n    def __splitfolders(self):\n        for i, line in tqdm(\n                enumerate(self.samples),\n                desc=\"Dataset creation\", total=self.length\n        ):\n            self.labels_counter = 0\n            identifier = line[\"id\"]\n            annotations = line[\"annotations\"]\n            paths_dict = self.__define_paths(i)\n\n            dst_image_path = self.__get_image_path(paths_dict, identifier)\n            dst_label_path = self.__get_label_path(paths_dict, identifier)\n\n            self.__copy_image(dst_image_path, identifier)\n            self.__copy_label(annotations, dst_label_path)\n\n            if self.labels_counter == 0:\n                os.remove(dst_image_path)\n                os.remove(dst_label_path)\n\n    def __count_dataset(self) -> dict:\n        train_images = len(os.listdir(os.path.join(self.train_dirpath, \"images\")))\n        train_labels = len(os.listdir(os.path.join(self.train_dirpath, \"labels\")))\n        val_images = len(os.listdir(os.path.join(self.val_dirpath, \"images\")))\n        val_labels = len(os.listdir(os.path.join(self.val_dirpath, \"labels\")))\n        return {\n            \"train_images\": train_images,\n            \"train_labels\": train_labels,\n            \"val_images\": val_images,\n            \"val_labels\": val_labels\n        }\n\n    @staticmethod\n    def __check_sanity(count_dict: dict) -> None:\n        assert count_dict[\"train_images\"] == count_dict[\"train_labels\"]\n        assert count_dict[\"val_images\"] == count_dict[\"val_labels\"]\n\n    def __finalizing(self, count_dict: dict) -> None:\n        assert os.path.exists(self.dataset_dirpath)\n\n        example_structure = [\n            \"dataset\",\n            \"train\", \"labels\", \"images\",\n            \"val\", \"labels\", \"images\"\n        ]\n\n        dir_bone = (\n            dirname.split(\"/\")[-1]\n            for dirname, _, filenames in os.walk(self.dataset_dirpath)\n            if dirname.split(\"/\")[-1] in example_structure\n        )\n\n        try:\n            print(\"\\n~ HuBMAP Dataset Structure ~\\n\")\n            print(\n            f\"\"\"\n          ├── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n            \"\"\"\n            )\n        except StopIteration as e:\n            print(e)\n        else:\n            print(Fore.GREEN + \"-> Success\")\n            print(Fore.GREEN + f\"Train dataset: {count_dict['train_images']}\\nVal dataset: {count_dict['val_images']}\")\n\n    def get_config(self) ->dict:\n        names = [\"blood_vessel\", \"glomerulus\", \"unsure\"]\n        return {\n            \"train\": str(self.train_dirpath),\n            \"val\": str(self.val_dirpath),\n            \"names\": [names[i] for i in self.classes]\n        }\n\n    @staticmethod\n    def display_config(config: dict) -> None:\n        print(Fore.BLACK + \"\\n~ HuBMAP Config Structure ~\\n\")\n        print(\n        f\"\"\"\n      │   │\n      │   ├── train\n      │   │   └── {config['train']}/images\n      │   │\n      │   │\n      │   ├── val\n      │   │   └── {config['val']}/images\n      │   │\n      │   │\n      │   ├── names\n      │   │   └── {' '.join(config['names'])}\n        \"\"\"\n        )\n        print(Fore.GREEN + \"-> Success\")\n        print(Fore.GREEN + f\"Number of classes: {len(config['names'])}\"\n                           f\"\\nClasses: {' '.join(config['names'])}\" \n              )\n\n    def write_config(self, config: dict) -> None:\n        with open(self.config_path, mode=\"w\") as f:\n            yaml.safe_dump(stream=f, data=config)\n\n    def __call__(self, train_size: float,\n                 classes: list[int, ...],\n                 make_config: bool = True,\n                 normalize: bool = True\n                ) -> None:\n        \n        self.train_size = train_size\n        self.classes = classes\n        self.normalize = normalize\n        \n        self.__define_splitratio()\n        self.__prepare_dirs()\n        self.__splitfolders()\n        count_dict = self.__count_dataset()\n        self.__check_sanity(count_dict)\n        self.__finalizing(count_dict)\n        \n        if make_config:\n            config = self.get_config()\n            self.write_config(config)\n            self.display_config(config)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:07:40.880915Z","iopub.execute_input":"2023-11-24T16:07:40.881254Z","iopub.status.idle":"2023-11-24T16:07:41.020279Z","shell.execute_reply.started":"2023-11-24T16:07:40.881221Z","shell.execute_reply":"2023-11-24T16:07:41.019433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco = COCODataset(\n    annotations_filepath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\",\n    images_dirpath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\",\n) ","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:07:41.021548Z","iopub.execute_input":"2023-11-24T16:07:41.021886Z","iopub.status.idle":"2023-11-24T16:07:45.253959Z","shell.execute_reply.started":"2023-11-24T16:07:41.021857Z","shell.execute_reply":"2023-11-24T16:07:45.252708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco(train_size=0.80, classes=[0, 1, 2])","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:07:45.25624Z","iopub.execute_input":"2023-11-24T16:07:45.256579Z","iopub.status.idle":"2023-11-24T16:08:30.501192Z","shell.execute_reply.started":"2023-11-24T16:07:45.25655Z","shell.execute_reply":"2023-11-24T16:08:30.500219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\nimport sys\nfrom colorama import Fore","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:08:30.502351Z","iopub.execute_input":"2023-11-24T16:08:30.502659Z","iopub.status.idle":"2023-11-24T16:08:30.50723Z","shell.execute_reply.started":"2023-11-24T16:08:30.502633Z","shell.execute_reply":"2023-11-24T16:08:30.506355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SetupPipline:\n    def __init__(self, display: bool = True):\n        self.pycocotools = self.__pycocotools()\n        self.ultralytics = self.__ultralytics()\n        \n    @staticmethod\n    def __ultralytics() -> str:\n        sys.path.append(\"/kaggle/input/hubmap-tools-ultralytics-and-pycocotools/ultralytics/ultralytics\") \n        return \"successfully\"\n        \n    @staticmethod\n    def __pycocotools() -> str:\n        if not os.path.exists(\"/kaggle/working/packages\"):\n            shutil.copytree(\"/kaggle/input/hubmap-tools-ultralytics-and-pycocotools/pycocotools/pycocotools\", \"/kaggle/working/packages\")\n            os.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n            os.system(\"python setup.py install\")\n            os.system(\"pip install . --no-index --find-links /kaggle/working/packages/\")\n            os.chdir(\"/kaggle/working\")\n            return \"successfully\"\n    \n    def display(self) -> None:\n        print(Fore.GREEN+f\"\\nPycocotools was installed {self.pycocotools}\")\n        print(f\"Ultralytics was installed {self.ultralytics}\"+Fore.WHITE)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:08:30.508609Z","iopub.execute_input":"2023-11-24T16:08:30.509025Z","iopub.status.idle":"2023-11-24T16:08:30.519416Z","shell.execute_reply.started":"2023-11-24T16:08:30.508994Z","shell.execute_reply":"2023-11-24T16:08:30.518573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipline = SetupPipline()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-24T16:08:30.520555Z","iopub.execute_input":"2023-11-24T16:08:30.520794Z","iopub.status.idle":"2023-11-24T16:09:17.597507Z","shell.execute_reply.started":"2023-11-24T16:08:30.520774Z","shell.execute_reply":"2023-11-24T16:09:17.596441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipline.display()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:09:17.598989Z","iopub.execute_input":"2023-11-24T16:09:17.599391Z","iopub.status.idle":"2023-11-24T16:09:17.60476Z","shell.execute_reply.started":"2023-11-24T16:09:17.599344Z","shell.execute_reply":"2023-11-24T16:09:17.60373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools import _mask as coco_mask \nfrom ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:09:17.609807Z","iopub.execute_input":"2023-11-24T16:09:17.610135Z","iopub.status.idle":"2023-11-24T16:09:24.68719Z","shell.execute_reply.started":"2023-11-24T16:09:17.610109Z","shell.execute_reply":"2023-11-24T16:09:24.686426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main():\n    model = YOLO(\"yolov8x-seg.pt\")\n    model.train(\n        # Project\n        project=\"HuBMAP\",\n        name=\"yolov8x-seg\",\n\n        # Random Seed parameters\n        deterministic=True,\n        seed=43,\n\n        # Data & model parameters\n        data=\"/kaggle/working/dataset/coco.yaml\", \n        save=True,\n        save_period=5,\n        pretrained=True,\n        imgsz=512,\n\n        # Training parameters\n        epochs=30,\n        batch=4,\n        workers=8,\n        val=True,\n        device=0,\n\n        # Optimization parameters\n        lr0=0.018,\n        patience=3,\n        optimizer=\"SGD\",\n        momentum=0.947,\n        weight_decay=0.0005,\n        close_mosaic=3,\n    )\n","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:09:24.688278Z","iopub.execute_input":"2023-11-24T16:09:24.688883Z","iopub.status.idle":"2023-11-24T16:09:24.695727Z","shell.execute_reply.started":"2023-11-24T16:09:24.688847Z","shell.execute_reply":"2023-11-24T16:09:24.694797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    main()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-24T16:09:24.696863Z","iopub.execute_input":"2023-11-24T16:09:24.697184Z","iopub.status.idle":"2023-11-24T16:30:40.015417Z","shell.execute_reply.started":"2023-11-24T16:09:24.69716Z","shell.execute_reply":"2023-11-24T16:30:40.014186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict\n\n","metadata":{"_kg_hide-output":false}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:30:40.017571Z","iopub.execute_input":"2023-11-24T16:30:40.018704Z","iopub.status.idle":"2023-11-24T16:30:40.026841Z","shell.execute_reply.started":"2023-11-24T16:30:40.018655Z","shell.execute_reply":"2023-11-24T16:30:40.025509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Choose any picture</b>\n</div>","metadata":{}},{"cell_type":"code","source":"dirlist = os.listdir(\"/kaggle/working/dataset/val/images\")\nprint(dirlist[:5])","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:30:40.029014Z","iopub.execute_input":"2023-11-24T16:30:40.029615Z","iopub.status.idle":"2023-11-24T16:30:40.541738Z","shell.execute_reply.started":"2023-11-24T16:30:40.029578Z","shell.execute_reply":"2023-11-24T16:30:40.540555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Displaying the image</b>\n</div>","metadata":{}},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\")\nhistory = model.predict(\"../working/dataset/val/images/ed6a92a9410c.tif\")[0]\nimage = history.plot()\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:30:40.543319Z","iopub.execute_input":"2023-11-24T16:30:40.54368Z","iopub.status.idle":"2023-11-24T16:30:41.993123Z","shell.execute_reply.started":"2023-11-24T16:30:40.543647Z","shell.execute_reply":"2023-11-24T16:30:41.992065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Result\n","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Losses & Recall & Precision & mAP</b>\n</div>","metadata":{}},{"cell_type":"code","source":"F1_curve = Image.open(\"/kaggle/working/HuBMAP/yolov8x-seg/results.png\")\nplt.figure(figsize=(15,20))\nplt.imshow(F1_curve)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:30:41.994667Z","iopub.execute_input":"2023-11-24T16:30:41.995065Z","iopub.status.idle":"2023-11-24T16:30:42.760791Z","shell.execute_reply.started":"2023-11-24T16:30:41.995027Z","shell.execute_reply":"2023-11-24T16:30:42.759788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Train Batch</b>\n</div>","metadata":{}},{"cell_type":"code","source":"P_curve = Image.open(\"/kaggle/working/HuBMAP/yolov8x-seg/train_batch0.jpg\")\nplt.imshow(P_curve)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T16:30:42.762011Z","iopub.execute_input":"2023-11-24T16:30:42.762289Z","iopub.status.idle":"2023-11-24T16:30:43.172104Z","shell.execute_reply.started":"2023-11-24T16:30:42.762266Z","shell.execute_reply":"2023-11-24T16:30:43.171082Z"},"trusted":true},"execution_count":null,"outputs":[]}]}