{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #abffd1; font-family:verdana; color: #003819; border: 2px #003819 solid\">\n    <b>Libraries for creating dataset</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# pip install clearml\n\n# %env CLEARML_WEB_HOST=https://app.clear.ml\n# %env CLEARML_API_HOST=https://api.clear.ml\n# %env CLEARML_FILES_HOST=https://files.clear.ml\n# %env CLEARML_API_ACCESS_KEY=EYSR99V3VZYP5AUTIJT3\n# %env CLEARML_API_SECRET_KEY=mC4uDuw0REYJaDDICNgSkyVDM1mtbXHT9jEgWxYziGR3JdU8V2\n\n# from clearml import Task\n# task = Task.init(project_name=\"test_1\", task_name=\"test_1_2\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:05.980186Z","iopub.execute_input":"2023-07-21T07:46:05.980531Z","iopub.status.idle":"2023-07-21T07:46:06.00551Z","shell.execute_reply.started":"2023-07-21T07:46:05.980501Z","shell.execute_reply":"2023-07-21T07:46:06.004757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import chain\nimport json\nimport os\nimport shutil\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore\nimport yaml\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:06.007185Z","iopub.execute_input":"2023-07-21T07:46:06.008674Z","iopub.status.idle":"2023-07-21T07:46:06.092139Z","shell.execute_reply.started":"2023-07-21T07:46:06.008624Z","shell.execute_reply":"2023-07-21T07:46:06.091161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #abffd1; font-family:verdana; color: #003819; border: 2px #003819 solid\">\n    <b>Class for creating dataset</b>\n</div>","metadata":{}},{"cell_type":"code","source":"class COCODataset:\n    def __init__(self, images_dirpath: str, annotations_filepath: str, length: int = 1633):\n        self.train_size = None\n        self.val_size = None\n        self.length = length\n        self.classes = None\n        self.labels_counter = None\n        self.normalize = None\n        \n        self.images_dirpath = images_dirpath\n        self.annotations_filepath = annotations_filepath\n        self.dataset_dirpath = os.path.join(os.getcwd(), \"dataset\")\n        self.train_dirpath =  os.path.join(self.dataset_dirpath, \"train\")\n        self.val_dirpath =  os.path.join(self.dataset_dirpath, \"val\")\n        self.config_path = os.path.join(self.dataset_dirpath, \"coco.yaml\")\n\n        self.samples = self.parse_jsonl(annotations_filepath)\n        self.classes_dict = {\n            \"blood_vessel\": 0,\n            \"glomerulus\": 1,\n            \"unsure\": 2,\n        }\n\n    def __prepare_dirs(self) -> None:\n        if not os.path.exists(self.dataset_dirpath):\n            os.makedirs(os.path.join(self.train_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.train_dirpath, \"labels\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"labels\"), exist_ok=True)\n        else:\n            raise RuntimeError(\"Dataset already exists!\")\n\n    def __define_splitratio(self) -> None:\n        self.train_size = round(self.length * self.train_size)\n        self.val_size = self.length - self.train_size\n        assert self.train_size + self.val_size == self.length\n\n    def parse_jsonl(self, path: str) -> list[dict, ...]:\n        with open(path, 'r') as json_file:\n            jsonl_samples = [\n                json.loads(line)\n                for line in tqdm(\n                    json_file, desc=\"Processing polygons\", total=self.length\n                )\n            ]\n        return jsonl_samples\n\n    def __define_paths(self, i: int) -> dict:\n        data_path = self.val_dirpath\n        if i < self.train_size:\n            data_path = self.train_dirpath\n        return {\n            \"images\": os.path.join(data_path, \"images\"),\n            \"labels\": os.path.join(data_path, \"labels\")\n        }\n\n    @staticmethod\n    def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"labels\"],\n            f\"{identifier}.txt\"\n        )\n\n    @staticmethod\n    def __get_image_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"images\"],\n            f\"{identifier}.tif\"\n        )\n\n    def __copy_image(self, dst_path: str, identifier: str) -> str:\n        shutil.copyfile(\n            os.path.join(self.images_dirpath, f\"{identifier}.tif\"),\n            dst_path\n        )\n\n    def __copy_label(self, annotations: list, dst_path: str) -> None:\n        with open(dst_path, \"w\") as file:\n            for annotation in annotations:\n                coordinates = annotation[\"coordinates\"][0]\n                label = self.classes_dict[annotation[\"type\"]]\n                if label in self.classes:\n                    if coordinates:\n                        if self.normalize:\n                            coordinates = np.array(coordinates) / 512.0                            \n                        coordinates = \" \".join(map(str, chain(*coordinates)))\n                        file.write(f\"{label} {coordinates}\\n\")\n                        self.labels_counter += 1\n\n    def __splitfolders(self):\n        for i, line in tqdm(\n                enumerate(self.samples),\n                desc=\"Dataset creation\", total=self.length\n        ):\n            self.labels_counter = 0\n            identifier = line[\"id\"]\n            annotations = line[\"annotations\"]\n            paths_dict = self.__define_paths(i)\n\n            dst_image_path = self.__get_image_path(paths_dict, identifier)\n            dst_label_path = self.__get_label_path(paths_dict, identifier)\n\n            self.__copy_image(dst_image_path, identifier)\n            self.__copy_label(annotations, dst_label_path)\n\n            if self.labels_counter == 0:\n                os.remove(dst_image_path)\n                os.remove(dst_label_path)\n\n    def __count_dataset(self) -> dict:\n        train_images = len(os.listdir(os.path.join(self.train_dirpath, \"images\")))\n        train_labels = len(os.listdir(os.path.join(self.train_dirpath, \"labels\")))\n        val_images = len(os.listdir(os.path.join(self.val_dirpath, \"images\")))\n        val_labels = len(os.listdir(os.path.join(self.val_dirpath, \"labels\")))\n        return {\n            \"train_images\": train_images,\n            \"train_labels\": train_labels,\n            \"val_images\": val_images,\n            \"val_labels\": val_labels\n        }\n\n    @staticmethod\n    def __check_sanity(count_dict: dict) -> None:\n        assert count_dict[\"train_images\"] == count_dict[\"train_labels\"]\n        assert count_dict[\"val_images\"] == count_dict[\"val_labels\"]\n\n    def __finalizing(self, count_dict: dict) -> None:\n        assert os.path.exists(self.dataset_dirpath)\n\n        example_structure = [\n            \"dataset\",\n            \"train\", \"labels\", \"images\",\n            \"val\", \"labels\", \"images\"\n        ]\n\n        dir_bone = (\n            dirname.split(\"/\")[-1]\n            for dirname, _, filenames in os.walk(self.dataset_dirpath)\n            if dirname.split(\"/\")[-1] in example_structure\n        )\n\n        try:\n            print(\"\\n~ HuBMAP Dataset Structure ~\\n\")\n            print(\n            f\"\"\"\n          ├── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n            \"\"\"\n            )\n        except StopIteration as e:\n            print(e)\n        else:\n            print(Fore.GREEN + \"-> Success\")\n            print(Fore.GREEN + f\"Train dataset: {count_dict['train_images']}\\nVal dataset: {count_dict['val_images']}\")\n\n    def get_config(self) ->dict:\n        names = [\"blood_vessel\", \"glomerulus\", \"unsure\"]\n        return {\n            \"train\": str(self.train_dirpath),\n            \"val\": str(self.val_dirpath),\n            \"names\": [names[i] for i in self.classes]\n        }\n\n    @staticmethod\n    def display_config(config: dict) -> None:\n        print(Fore.BLACK + \"\\n~ HuBMAP Config Structure ~\\n\")\n        print(\n        f\"\"\"\n      │   │\n      │   ├── train\n      │   │   └── {config['train']}/images\n      │   │\n      │   │\n      │   ├── val\n      │   │   └── {config['val']}/images\n      │   │\n      │   │\n      │   ├── names\n      │   │   └── {' '.join(config['names'])}\n        \"\"\"\n        )\n        print(Fore.GREEN + \"-> Success\")\n        print(Fore.GREEN + f\"Number of classes: {len(config['names'])}\"\n                           f\"\\nClasses: {' '.join(config['names'])}\" \n              )\n\n    def write_config(self, config: dict) -> None:\n        with open(self.config_path, mode=\"w\") as f:\n            yaml.safe_dump(stream=f, data=config)\n\n    def __call__(self, train_size: float,\n                 classes: list[int, ...],\n                 make_config: bool = True,\n                 normalize: bool = True\n                ) -> None:\n        \n        self.train_size = train_size\n        self.classes = classes\n        self.normalize = normalize\n        \n        self.__define_splitratio()\n        self.__prepare_dirs()\n        self.__splitfolders()\n        count_dict = self.__count_dataset()\n        self.__check_sanity(count_dict)\n        self.__finalizing(count_dict)\n        \n        if make_config:\n            config = self.get_config()\n            self.write_config(config)\n            self.display_config(config)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:06.094704Z","iopub.execute_input":"2023-07-21T07:46:06.095029Z","iopub.status.idle":"2023-07-21T07:46:06.128446Z","shell.execute_reply.started":"2023-07-21T07:46:06.095005Z","shell.execute_reply":"2023-07-21T07:46:06.127288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #abffd1; font-family:verdana; color: #003819; border: 2px #003819 solid\">\n    <b>Let's fire up the data 🔥</b>\n</div>","metadata":{}},{"cell_type":"code","source":"coco = COCODataset(\n    annotations_filepath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\",\n    images_dirpath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\",\n) ","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:06.129602Z","iopub.execute_input":"2023-07-21T07:46:06.129886Z","iopub.status.idle":"2023-07-21T07:46:09.175826Z","shell.execute_reply.started":"2023-07-21T07:46:06.129863Z","shell.execute_reply":"2023-07-21T07:46:09.174692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco(train_size=0.80, classes=[0, 1, 2])","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:09.178399Z","iopub.execute_input":"2023-07-21T07:46:09.178719Z","iopub.status.idle":"2023-07-21T07:46:38.918221Z","shell.execute_reply.started":"2023-07-21T07:46:09.178695Z","shell.execute_reply":"2023-07-21T07:46:38.916919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #f2ffb0; font-family:verdana; color: #3c450e; border: 2px #3c450e solid\">\n    <b>And how can I use the right tools if the internet is banned in the competition? 🙁</b>\n    <br>In the competition, many have difficulty installing the necessary tools and using them, because the Internet is prohibited, but it is prohibited only so that the test data is not stolen (they are uploaded to the test folder during the submission of the result). This means that you can use the tools you want, but you must add them as data. You can find my ultralytics and pycocotools dataset here. Importantly, the Internet is still present in this notebook to load the model, however, after training it, you will save the model, add it to the data and easily use it for forecasting without the Internet.<br>\n</div>","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os\nimport sys\nfrom colorama import Fore","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:38.920028Z","iopub.execute_input":"2023-07-21T07:46:38.920373Z","iopub.status.idle":"2023-07-21T07:46:38.925393Z","shell.execute_reply.started":"2023-07-21T07:46:38.920347Z","shell.execute_reply":"2023-07-21T07:46:38.924474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SetupPipline:\n    def __init__(self, display: bool = True):\n        self.pycocotools = self.__pycocotools()\n        self.ultralytics = self.__ultralytics()\n        \n    @staticmethod\n    def __ultralytics() -> str:\n        sys.path.append(\"/kaggle/input/hubmap-tools-ultralytics-and-pycocotools/ultralytics/ultralytics\")\n#         sys.path.append(\"/kaggle/input/yolov8\") \n#         sys.path.append(\"/kaggle/input/used-model/ultralytics\") \n        return \"successfully\"\n        \n    @staticmethod\n    def __pycocotools() -> str:\n        if not os.path.exists(\"/kaggle/working/packages\"):\n            shutil.copytree(\"/kaggle/input/hubmap-tools-ultralytics-and-pycocotools/pycocotools/pycocotools\", \"/kaggle/working/packages\")\n            os.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n            os.system(\"python setup.py install\")\n            os.system(\"pip install . --no-index --find-links /kaggle/working/packages/\")\n            os.chdir(\"/kaggle/working\")\n            return \"successfully\"\n    \n    def display(self) -> None:\n        print(Fore.GREEN+f\"\\nPycocotools was installed {self.pycocotools}\")\n        print(f\"Ultralytics was installed {self.ultralytics}\"+Fore.WHITE)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:46:38.927007Z","iopub.execute_input":"2023-07-21T07:46:38.927455Z","iopub.status.idle":"2023-07-21T07:46:38.938581Z","shell.execute_reply.started":"2023-07-21T07:46:38.927424Z","shell.execute_reply":"2023-07-21T07:46:38.937682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipline = SetupPipline()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-21T07:46:38.93969Z","iopub.execute_input":"2023-07-21T07:46:38.940405Z","iopub.status.idle":"2023-07-21T07:47:23.257461Z","shell.execute_reply.started":"2023-07-21T07:46:38.940373Z","shell.execute_reply":"2023-07-21T07:47:23.256395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipline.display()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:47:23.258353Z","iopub.execute_input":"2023-07-21T07:47:23.258609Z","iopub.status.idle":"2023-07-21T07:47:23.263703Z","shell.execute_reply.started":"2023-07-21T07:47:23.258589Z","shell.execute_reply":"2023-07-21T07:47:23.262556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools import _mask as coco_mask \nfrom ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:47:23.267861Z","iopub.execute_input":"2023-07-21T07:47:23.268212Z","iopub.status.idle":"2023-07-21T07:47:28.553179Z","shell.execute_reply.started":"2023-07-21T07:47:23.268183Z","shell.execute_reply":"2023-07-21T07:47:28.552013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training\n<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #eec4ff; font-family:verdana; color: #511b66; border: 2px #511b66 solid\">\n    <b>Finally, we can start learning 🥳</b>\n    <br>From this moment begins our greedy search. We hypothesize hypotheses, select hypermaparmeters, train, and then check whether the hypothesis was correct using the resulting metrics. This stage is one of the most interesting. Offhand, I can say that the best results can be obtained on a pre-trained model with the parameters below (well + - still try to experiment)<br>\n</div>","metadata":{}},{"cell_type":"code","source":"\ndef main():\n    model = YOLO(\"yolov8x-seg.pt\")\n    model.train(\n        # Project\n        project=\"HuBMAP\",\n        name=\"yolov8x-seg\",\n\n        # Random Seed parameters\n        deterministic=True,\n        seed=43,\n\n        # Data & model parameters\n        data=\"/kaggle/working/dataset/coco.yaml\", \n        save=True,\n        save_period=5,\n        pretrained=True,\n        imgsz=512,\n\n        # Training parameters\n        epochs=20,\n        batch=4,\n        workers=8,\n        val=True,\n#         device=0,\n\n        # Optimization parameters\n        lr0=0.018,\n        patience=3,\n        optimizer=\"SGD\",\n        momentum=0.947,\n        weight_decay=0.0005,\n        close_mosaic=3,\n    )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:47:28.554722Z","iopub.execute_input":"2023-07-21T07:47:28.55536Z","iopub.status.idle":"2023-07-21T07:47:28.56286Z","shell.execute_reply.started":"2023-07-21T07:47:28.555314Z","shell.execute_reply":"2023-07-21T07:47:28.561827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #eec4ff; font-family:verdana; color: #511b66; border: 2px #511b66 solid\">\n    <b>Please follow the <a href=https://wandb.ai/site>link</a>. Sign up and then paste your API key into the box that pops up below. Do not disclose your key to anyone. Here it is required so that you can track the training</b>\n</div>","metadata":{}},{"cell_type":"code","source":"if __name__ == '__main__':\n    !wandb off\n    main()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-21T07:47:28.564431Z","iopub.execute_input":"2023-07-21T07:47:28.564902Z","iopub.status.idle":"2023-07-21T07:50:34.704227Z","shell.execute_reply.started":"2023-07-21T07:47:28.564869Z","shell.execute_reply":"2023-07-21T07:50:34.701851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict\n\n<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n<b>The training has come to an end and now it is time to look at the results of the work of the neural network</b>\n<ul style=\"font-size:20px; font-family:verdana; line-height: 1.7em\">\n    <li>Choose any picture</li>\n    <li>Displaying the image</li>\n</ul>\n</div>","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.705413Z","iopub.status.idle":"2023-07-21T07:50:34.706111Z","shell.execute_reply.started":"2023-07-21T07:50:34.705952Z","shell.execute_reply":"2023-07-21T07:50:34.705969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Choose any picture</b>\n</div>","metadata":{}},{"cell_type":"code","source":"dirlist = os.listdir(\"/kaggle/working/dataset/val/images\")\nprint(dirlist[:5])","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.706997Z","iopub.status.idle":"2023-07-21T07:50:34.707568Z","shell.execute_reply.started":"2023-07-21T07:50:34.707414Z","shell.execute_reply":"2023-07-21T07:50:34.70743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Displaying the image</b>\n</div>","metadata":{}},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\")\nhistory = model.predict(\"../working/dataset/val/images/ed6a92a9410c.tif\")[0]\nimage = history.plot()\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.709042Z","iopub.status.idle":"2023-07-21T07:50:34.709344Z","shell.execute_reply.started":"2023-07-21T07:50:34.709188Z","shell.execute_reply":"2023-07-21T07:50:34.709202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's build the graphs ","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Losses & Recall & Precision & mAP</b>\n</div>","metadata":{}},{"cell_type":"code","source":"F1_curve = Image.open(\"/kaggle/working/HuBMAP/yolov8x-seg/results.png\")\nplt.figure(figsize=(15,20))\nplt.imshow(F1_curve)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.710517Z","iopub.status.idle":"2023-07-21T07:50:34.711141Z","shell.execute_reply.started":"2023-07-21T07:50:34.710976Z","shell.execute_reply":"2023-07-21T07:50:34.710992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; font-family:verdana;\">\n    <b>Train Batch</b>\n</div>","metadata":{}},{"cell_type":"code","source":"P_curve = Image.open(\"/kaggle/working/HuBMAP/yolov8x-seg/train_batch5561.jpg\")\nplt.imshow(P_curve)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.712884Z","iopub.status.idle":"2023-07-21T07:50:34.713394Z","shell.execute_reply.started":"2023-07-21T07:50:34.713164Z","shell.execute_reply":"2023-07-21T07:50:34.713187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #ffa340; font-family:verdana; color: #663500; border: 2px #663500 solid\">\n    <b>What about submission? 😸</b>\n    <br>We have come a long way, done data analysis, trained the model, but what about the submission? We want to send the results, right? Well, I had difficulties with this, and therefore, in order to make life easier for myself, it is possible to help someone, I wrote several classes that allow you to quickly submit without delving into the difficulties that I encountered.<br>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #ffc98f; font-family:verdana; color: #b86e1f; border: 2px #b86e1f solid\">\n    <b>Encode Binary Mask</b>\n</div>","metadata":{}},{"cell_type":"code","source":"import base64\nimport numpy as np\nimport torch\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\nimport pandas as pd\nimport torchvision.transforms as T\nfrom ultralytics import YOLO\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.71514Z","iopub.status.idle":"2023-07-21T07:50:34.715587Z","shell.execute_reply.started":"2023-07-21T07:50:34.715375Z","shell.execute_reply":"2023-07-21T07:50:34.715394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EncodeBinaryMask:\n    @staticmethod\n    def __checking_mask(mask: np.ndarray) -> np.ndarray:\n        if mask.dtype != np.bool:\n            raise ValueError(\n                \"expects a binary mask, received dtype == %s\" %\n                mask.dtype\n            )\n        return mask\n\n    @staticmethod\n    def __convert_mask(mask: np.ndarray):\n        mask_to_encode = mask.astype(np.uint8)\n        mask_to_encode = np.asfortranarray(mask_to_encode)\n        return mask_to_encode\n\n    @staticmethod\n    def __compress_encode(encoded_mask) -> t.Text:\n        binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n        base64_str = base64.b64encode(binary_str)\n        return base64_str\n\n    def __call__(self, mask: np.ndarray) -> t.Text:\n        mask = self.__checking_mask(mask)\n        mask_to_encode = self.__convert_mask(mask)\n        encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n        base64_str = self.__compress_encode(encoded_mask)\n        return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.717949Z","iopub.status.idle":"2023-07-21T07:50:34.718713Z","shell.execute_reply.started":"2023-07-21T07:50:34.718442Z","shell.execute_reply":"2023-07-21T07:50:34.718467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #ffc98f; font-family:verdana; color: #b86e1f; border: 2px #b86e1f solid\">\n    <b>Submission</b>\n</div>","metadata":{}},{"cell_type":"code","source":"class Submission:\n    def __init__(self, dirpath: str, model: torch.nn.Module):\n        self.__eval_transforms = self.get_transforms()\n        self.__model = model\n        self.__encoder = EncodeBinaryMask()\n        self.__dirpath = dirpath\n        self.__filenames = os.listdir(dirpath)\n        self.height = 512\n        self.width = 512\n        \n        self.__submission_dict = {\n            \"id\": [],\n            \"height\": [],\n            \"width\": [],\n            \"prediction_string\": []\n        }\n        \n        self.submission = None\n    \n    @staticmethod\n    def get_transforms():\n        return T.Compose([\n            T.ToTensor(),\n            T.Resize(size=(512, 512)),\n            T.Normalize(mean=[0.485, 0.456, 0.406],\n                        std=[0.229, 0.224, 0.225])\n        ])\n\n    def __len__(self):\n        return len(self.__filenames)\n\n    def __get_columns(self) -> None:\n        for filename in self.__filenames:\n            path = self.__get_image_path(filename)\n            masks = self.__forward(path)\n            identifier, height, width, prediction_string = self.__get_cells(filename, masks)\n            self.__update_columns(identifier, height, width, prediction_string)\n\n    def __update_columns(self, identifier: str, height: int, width: int, prediction_string: str) -> None:\n        self.__submission_dict[\"id\"].append(identifier)\n        self.__submission_dict[\"height\"].append(height)\n        self.__submission_dict[\"width\"].append(width)\n        self.__submission_dict[\"prediction_string\"].append(prediction_string)\n\n    def __get_cells(self, filename: str, masks: list):\n        prediction_string = \"\"\n        prediction_string = self.__get_prediction_string(masks, prediction_string)\n        identifier = filename.split(\".\")[0]\n        return identifier, self.height, self.width, prediction_string\n\n    def __get_prediction_string(self, masks: list, prediction_string: str) -> str:\n        if masks:\n            for outputs in masks:\n                mask = outputs[\"mask\"]\n                mask = np.where(mask > 0.5, 1, 0).astype(np.bool)\n                base64_str = self.__encoder(mask)\n                confidence = outputs[\"confidence\"]\n                prediction_string += f\"0 {confidence} {base64_str.decode('utf-8')} \"\n        else:\n            return \"\"\n        return prediction_string\n\n    def __get_image_path(self, filename: str) -> str:\n        return os.path.join(\n            self.__dirpath, filename\n        )\n\n    def __get_image(self, path: str) -> torch.Tensor:\n        image = Image.open(path)\n        image = np.asarray(image)\n        image = self.__eval_transforms(image)\n        return image\n\n    def __forward(self, image: torch.tensor) -> list:\n        masks = self.__model(image) \n        return masks \n\n    def submit(self) -> None:\n        if not self.submission:\n            self.__get_columns()\n            self.submission = pd.DataFrame(self.__submission_dict)\n            self.submission = self.submission.set_index('id')\n            self.submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.720722Z","iopub.status.idle":"2023-07-21T07:50:34.721232Z","shell.execute_reply.started":"2023-07-21T07:50:34.72099Z","shell.execute_reply":"2023-07-21T07:50:34.721012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:20px; background-color: #ffc98f; font-family:verdana; color: #b86e1f; border: 2px #b86e1f solid\">\n    <b>Wrapper class for YOLO model</b>\n</div>","metadata":{}},{"cell_type":"code","source":"class BestYolo:\n    def __init__(self, conf: float = 0.05):\n        self.model_path = \"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\"\n        self.model = self.get_model()\n        self.conf = conf\n    \n    def get_model(self) -> YOLO:\n        return YOLO(self.model_path)\n    \n    def __call__(self, source) -> list[dict, ...]:\n        sublist = []\n        result = self.model(source)[0]\n        if result.masks:\n            for i in range(len(result.masks.data)):\n                conf = round(float(result.boxes.conf[i]), 2)\n                mask = np.expand_dims(result.masks.data[i].cpu().numpy(), axis=0).transpose(1,2,0)\n            \n                if int(result.boxes.cls[i]) == 0 and conf >= self.conf:\n                    sublist.append({\"mask\": mask, \"confidence\": conf})\n                else:\n                    continue\n            return sublist\n        else:\n            return None","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.72235Z","iopub.status.idle":"2023-07-21T07:50:34.722789Z","shell.execute_reply.started":"2023-07-21T07:50:34.722581Z","shell.execute_reply":"2023-07-21T07:50:34.722601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"__TEST_PATH = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\nmodel = BestYolo()\nsub = Submission(dirpath=__TEST_PATH, model=model)\nsub.submit()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.724713Z","iopub.status.idle":"2023-07-21T07:50:34.725181Z","shell.execute_reply.started":"2023-07-21T07:50:34.724952Z","shell.execute_reply":"2023-07-21T07:50:34.724972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.726875Z","iopub.status.idle":"2023-07-21T07:50:34.727309Z","shell.execute_reply.started":"2023-07-21T07:50:34.727086Z","shell.execute_reply":"2023-07-21T07:50:34.727105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport csv\nimport json\nimport os\nimport shutil\n\n# 1. 读取CSV文件，筛选出dataset类别为3的图片ID\nselected_image_ids = []\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv', 'r') as csvfile:\n    csvreader = csv.reader(csvfile)\n    next(csvreader)\n    for row in csvreader:\n        image_id, dataset_category = row[0], int(row[2])\n        if dataset_category == 3:\n            selected_image_ids.append(image_id)\n\n# 2. 提取图片，准备用于预测\n# 源文件夹路径\nsource_folder = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\n\n# 目标文件夹路径，用于存储提取出来的图片\ndestination_folder = \"/kaggle/working/dataset3\"\n\n# 创建目标文件夹\nos.makedirs(destination_folder, exist_ok=True)\n\n# 遍历源文件夹中的所有文件\nfor filename in os.listdir(source_folder):\n    # 获取文件的ID，假设文件名的格式是\"image_id.jpg\"\n    image_id = filename.split(\".\")[0]\n\n    # 如果文件的ID在selected_image_ids中，则将文件复制到目标文件夹\n    if image_id in selected_image_ids:\n        source_file = os.path.join(source_folder, filename)\n        destination_file = os.path.join(destination_folder, filename)\n        shutil.copyfile(source_file, destination_file)\n        \n       \nprint(\"Dataset3 set\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.72877Z","iopub.status.idle":"2023-07-21T07:50:34.729207Z","shell.execute_reply.started":"2023-07-21T07:50:34.728996Z","shell.execute_reply":"2023-07-21T07:50:34.729015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"txt to json","metadata":{}},{"cell_type":"code","source":"import os\nfrom PIL import Image\n\n# 初始化一个列表用于存储图片\nselected_images = []\n\n# 遍历源文件夹中的所有文件\nfor filename in os.listdir(\"/kaggle/working/dataset3\"):\n    # 假设只处理JPEG和PNG格式的图片文件，你可以根据实际情况添加其他格式\n    image_path = os.path.join(source_folder, filename)\n\n    # 使用Pillow库打开图像文件\n    try:\n        image = Image.open(image_path)\n        selected_images.append(image)\n        \n    except Exception as e:\n        print(f\"无法读取图片 {filename}，错误信息：{e}\")\n\nprint(\"所有图片已读取并放入selected_images列表\")\n\none_third_size = len(selected_images) // 3\nfirst_third = selected_images[:one_third_size]\nsecond_third = selected_images[one_third_size:2*one_third_size]\nthird_third = selected_images[2*one_third_size:]\n\n\n","metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"scrolled":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.730886Z","iopub.status.idle":"2023-07-21T07:50:34.731325Z","shell.execute_reply.started":"2023-07-21T07:50:34.731097Z","shell.execute_reply":"2023-07-21T07:50:34.731118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\")\nfor image in first_third:\n    # 假设模型的预测函数是predict(image)，返回预测结果，格式类似于示例中的detections\n    model.predict(image,save_txt = True)\nprint(\"predict first down\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.732567Z","iopub.status.idle":"2023-07-21T07:50:34.732988Z","shell.execute_reply.started":"2023-07-21T07:50:34.732766Z","shell.execute_reply":"2023-07-21T07:50:34.732785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\")\nfor image in second_third:\n    # 假设模型的预测函数是predict(image)，返回预测结果，格式类似于示例中的detections\n    model.predict(image,save_txt = True)\nprint(\"predict second down\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.735092Z","iopub.status.idle":"2023-07-21T07:50:34.735541Z","shell.execute_reply.started":"2023-07-21T07:50:34.735327Z","shell.execute_reply":"2023-07-21T07:50:34.735346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/HuBMAP/yolov8x-seg/weights/best.pt\")\nfor image in third_third:\n    # 假设模型的预测函数是predict(image)，返回预测结果，格式类似于示例中的detections\n    model.predict(image,save_txt = True)\nprint(\"predict third down\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.737106Z","iopub.status.idle":"2023-07-21T07:50:34.737519Z","shell.execute_reply.started":"2023-07-21T07:50:34.737318Z","shell.execute_reply":"2023-07-21T07:50:34.737337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport os\n\ndef extract_file_name(txt_file_path):\n    file_name = os.path.basename(txt_file_path)\n    file_name = os.path.splitext(file_name)[0]\n    return file_name\n\ndef parse_yolo_txt_to_json(yolo_txt,image_id):\n    annotations = []\n    lines = yolo_txt.strip().split(\"\\n\")\n    for line in lines:\n        parts = line.strip().split()\n        class_id = int(parts[0])\n        coordinates = [(float(parts[i]), float(parts[i + 1])) for i in range(1, len(parts), 2)]\n        \n        # Assuming class_id 0 is for blood_vessel, 1 for glomerulus, and 2 for unsure\n        annotation_type = \"blood_vessel\" if class_id == 0 else \"glomerulus\" if class_id == 1 else \"unsure\"\n        \n        annotation = {\n            \"type\": annotation_type,\n            \"coordinates\": coordinates\n        }\n        annotations.append(annotation)\n\n    json_data = {\n        \"id\": image_id,\n        \"annotations\": annotations\n    }\n\n    return json_data\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.738547Z","iopub.status.idle":"2023-07-21T07:50:34.738959Z","shell.execute_reply.started":"2023-07-21T07:50:34.738737Z","shell.execute_reply":"2023-07-21T07:50:34.738755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Directory containing the YOLO txt files for multiple images\nyolo_txt_directory1 = '/kaggle/working/runs/segment/predict/labels'\nyolo_txt_directory2 = '/kaggle/working/runs/segment/predict2/labels'\nyolo_txt_directory3 = '/kaggle/working/runs/segment/predict3/labels'\n# Output JSON file to store annotations for all images\noutput_json_file = '/kaggle/working/output_dataset3.json'\n\nwith open(output_json_file, 'w') as outfile:\n    pass\n\ndef append_to_json(yolo_txt_directory, output_json_file):\n    for txt_file in os.listdir(yolo_txt_directory):\n        if txt_file.endswith(\".txt\"):\n            txt_file_path = os.path.join(yolo_txt_directory, txt_file)\n            image_id = extract_file_name(txt_file)\n\n            with open(txt_file_path, 'r') as file:\n                yolo_txt = file.read()\n\n            json_data = parse_yolo_txt_to_json(yolo_txt, image_id)\n            \n            # Append the JSON data to the output JSON file\n            with open(output_json_file, 'a') as outfile:\n                json.dump(json_data, outfile)\n                outfile.write(\"\\n\")\n\nappend_to_json(yolo_txt_directory1, output_json_file)\nappend_to_json(yolo_txt_directory2, output_json_file)\nappend_to_json(yolo_txt_directory3, output_json_file)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-21T07:50:34.741459Z","iopub.status.idle":"2023-07-21T07:50:34.741891Z","shell.execute_reply.started":"2023-07-21T07:50:34.741658Z","shell.execute_reply":"2023-07-21T07:50:34.741677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\ndef convert_coordinates_to_int(json_data):\n    for sample in json_data:\n        for annotation in sample[\"annotations\"]:\n            coordinates = annotation[\"coordinates\"]\n            coordinates_int = [[int(coord * 512) for coord in point] for point in coordinates]\n            annotation[\"coordinates\"] = coordinates_int\n\ndef process_json_file(input_json_file, output_json_file):\n    with open(input_json_file, 'r') as infile:\n        json_lines = infile.readlines()\n\n    json_data_list = []\n    for line in json_lines:\n        json_data = json.loads(line.strip())\n        json_data_list.append(json_data)\n\n    convert_coordinates_to_int(json_data_list)\n\n    with open(output_json_file, 'w') as outfile:\n        for sample in json_data_list:\n            json.dump(sample, outfile)\n            outfile.write(\"\\n\")\n\n\ninput_json_file = \"/kaggle/working/output_dataset3.json\"\noutput_json_file = \"/kaggle/working/output_dataset3_int.json\"\n\nprocess_json_file(input_json_file, output_json_file)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.743765Z","iopub.status.idle":"2023-07-21T07:50:34.744238Z","shell.execute_reply.started":"2023-07-21T07:50:34.744019Z","shell.execute_reply":"2023-07-21T07:50:34.744038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"读取图片做预测","metadata":{}},{"cell_type":"code","source":"class COCODataset_dataset3:\n    def __init__(self, images_dirpath: str, annotations_filepath: str, length: int = 5100):\n        self.train_size = None\n        self.val_size = None\n        self.length = length\n        self.classes = None\n        self.labels_counter = None\n        self.normalize = None\n        \n        self.images_dirpath = images_dirpath\n        self.annotations_filepath = annotations_filepath\n        self.dataset_dirpath = os.path.join(os.getcwd(), \"dataset\")\n        self.train_dirpath =  os.path.join(self.dataset_dirpath, \"train\")\n        self.val_dirpath =  os.path.join(self.dataset_dirpath, \"val\")\n        self.config_path = os.path.join(self.dataset_dirpath, \"coco.yaml\")\n\n        self.samples = self.parse_jsonl(annotations_filepath)\n        self.classes_dict = {\n            \"blood_vessel\": 0,\n            \"glomerulus\": 1,\n            \"unsure\": 2,\n        }\n\n    def __prepare_dirs(self) -> None:\n        if not os.path.exists(self.dataset_dirpath):\n            os.makedirs(os.path.join(self.train_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.train_dirpath, \"labels\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"images\"), exist_ok=True)\n            os.makedirs(os.path.join(self.val_dirpath, \"labels\"), exist_ok=True)\n        else:\n            raise RuntimeError(\"Dataset already exists!\")\n\n    def __define_splitratio(self) -> None:\n        self.train_size = round(self.length * self.train_size)\n        self.val_size = self.length - self.train_size\n        assert self.train_size + self.val_size == self.length\n\n    def parse_jsonl(self, path: str) -> list[dict, ...]:\n        with open(path, 'r') as json_file:\n            jsonl_samples = [\n                json.loads(line)\n                for line in tqdm(\n                    json_file, desc=\"Processing polygons\", total=self.length\n                )\n            ]\n        return jsonl_samples\n\n    def __define_paths(self, i: int) -> dict:\n        data_path = self.val_dirpath\n        if i < self.train_size:\n            data_path = self.train_dirpath\n        return {\n            \"images\": os.path.join(data_path, \"images\"),\n            \"labels\": os.path.join(data_path, \"labels\")\n        }\n\n    @staticmethod\n    def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"labels\"],\n            f\"{identifier}.txt\"\n        )\n\n    @staticmethod\n    def __get_image_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"images\"],\n            f\"{identifier}.tif\"\n        )\n\n    def __copy_image(self, dst_path: str, identifier: str) -> str:\n        shutil.copyfile(\n            os.path.join(self.images_dirpath, f\"{identifier}.tif\"),\n            dst_path\n        )\n\n    def __copy_label(self, annotations: list, dst_path: str) -> None:\n        with open(dst_path, \"w\") as file:\n            for annotation in annotations:\n                coordinates = annotation[\"coordinates\"][0]\n                label = self.classes_dict[annotation[\"type\"]]\n                if label in self.classes:\n                    if coordinates:\n                        if self.normalize:\n                            coordinates = np.array(coordinates) / 512.0                            \n#                         coordinates = \" \".join(map(str, chain(*coordinates)))\n                        coordinates = \" \".join(map(str, np.array(coordinates).flatten()))\n                        file.write(f\"{label} {coordinates}\\n\")\n                        self.labels_counter += 1\n\n    def __splitfolders(self):\n        for i, line in tqdm(\n                enumerate(self.samples),\n                desc=\"Dataset creation\", total=self.length\n        ):\n            self.labels_counter = 0\n            identifier = line[\"id\"]\n            annotations = line[\"annotations\"]\n            paths_dict = self.__define_paths(i)\n\n            dst_image_path = self.__get_image_path(paths_dict, identifier)\n            dst_label_path = self.__get_label_path(paths_dict, identifier)\n\n            self.__copy_image(dst_image_path, identifier)\n            self.__copy_label(annotations, dst_label_path)\n\n            if self.labels_counter == 0:\n                os.remove(dst_image_path)\n                os.remove(dst_label_path)\n\n    def __count_dataset(self) -> dict:\n        train_images = len(os.listdir(os.path.join(self.train_dirpath, \"images\")))\n        train_labels = len(os.listdir(os.path.join(self.train_dirpath, \"labels\")))\n        val_images = len(os.listdir(os.path.join(self.val_dirpath, \"images\")))\n        val_labels = len(os.listdir(os.path.join(self.val_dirpath, \"labels\")))\n        return {\n            \"train_images\": train_images,\n            \"train_labels\": train_labels,\n            \"val_images\": val_images,\n            \"val_labels\": val_labels\n        }\n\n    @staticmethod\n    def __check_sanity(count_dict: dict) -> None:\n        assert count_dict[\"train_images\"] == count_dict[\"train_labels\"]\n        assert count_dict[\"val_images\"] == count_dict[\"val_labels\"]\n\n    def __finalizing(self, count_dict: dict) -> None:\n        assert os.path.exists(self.dataset_dirpath)\n\n        example_structure = [\n            \"dataset\",\n            \"train\", \"labels\", \"images\",\n            \"val\", \"labels\", \"images\"\n        ]\n\n        dir_bone = (\n            dirname.split(\"/\")[-1]\n            for dirname, _, filenames in os.walk(self.dataset_dirpath)\n            if dirname.split(\"/\")[-1] in example_structure\n        )\n\n        try:\n            print(\"\\n~ HuBMAP Dataset Structure ~\\n\")\n            print(\n            f\"\"\"\n          ├── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │\n          │   ├── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n          │   │   └── {next(dir_bone)}\n            \"\"\"\n            )\n        except StopIteration as e:\n            print(e)\n        else:\n            print(Fore.GREEN + \"-> Success\")\n            print(Fore.GREEN + f\"Train dataset: {count_dict['train_images']}\\nVal dataset: {count_dict['val_images']}\")\n\n    def get_config(self) ->dict:\n        names = [\"blood_vessel\", \"glomerulus\", \"unsure\"]\n        return {\n            \"train\": str(self.train_dirpath),\n            \"val\": str(self.val_dirpath),\n            \"names\": [names[i] for i in self.classes]\n        }\n\n    @staticmethod\n    def display_config(config: dict) -> None:\n        print(Fore.BLACK + \"\\n~ HuBMAP Config Structure ~\\n\")\n        print(\n        f\"\"\"\n      │   │\n      │   ├── train\n      │   │   └── {config['train']}/images\n      │   │\n      │   │\n      │   ├── val\n      │   │   └── {config['val']}/images\n      │   │\n      │   │\n      │   ├── names\n      │   │   └── {' '.join(config['names'])}\n        \"\"\"\n        )\n        print(Fore.GREEN + \"-> Success\")\n        print(Fore.GREEN + f\"Number of classes: {len(config['names'])}\"\n                           f\"\\nClasses: {' '.join(config['names'])}\" \n              )\n\n    def write_config(self, config: dict) -> None:\n        with open(self.config_path, mode=\"w\") as f:\n            yaml.safe_dump(stream=f, data=config)\n\n    def __call__(self, train_size: float,\n                 classes: list[int, ...],\n                 make_config: bool = True,\n                 normalize: bool = True\n                ) -> None:\n        \n        self.train_size = train_size\n        self.classes = classes\n        self.normalize = normalize\n        \n        self.__define_splitratio()\n        self.__prepare_dirs()\n        self.__splitfolders()\n        count_dict = self.__count_dataset()\n        self.__check_sanity(count_dict)\n        self.__finalizing(count_dict)\n        \n        if make_config:\n            config = self.get_config()\n            self.write_config(config)\n            self.display_config(config)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.745643Z","iopub.status.idle":"2023-07-21T07:50:34.746429Z","shell.execute_reply.started":"2023-07-21T07:50:34.746215Z","shell.execute_reply":"2023-07-21T07:50:34.746236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco_dataset3 = COCODataset_dataset3(\n    annotations_filepath=\"/kaggle/working/output_dataset3_int.json\",\n    images_dirpath=\"/kaggle/working/dataset3\",\n) ","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.74826Z","iopub.status.idle":"2023-07-21T07:50:34.748558Z","shell.execute_reply.started":"2023-07-21T07:50:34.748414Z","shell.execute_reply":"2023-07-21T07:50:34.748428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\nfolder_path = \"/kaggle/working/dataset\"  \nshutil.rmtree(folder_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.752455Z","iopub.status.idle":"2023-07-21T07:50:34.752925Z","shell.execute_reply.started":"2023-07-21T07:50:34.752705Z","shell.execute_reply":"2023-07-21T07:50:34.752722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco_dataset3(train_size=0.80, classes=[0, 1, 2])","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.75458Z","iopub.status.idle":"2023-07-21T07:50:34.755102Z","shell.execute_reply.started":"2023-07-21T07:50:34.754882Z","shell.execute_reply":"2023-07-21T07:50:34.754902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main():\n    model = YOLO(\"/kaggle/working/yolov8x-seg.pt\")\n    model.train(\n        # Project\n        project=\"HuBMAP\",\n        name=\"yolov8x-seg-dateset3\",\n\n        # Random Seed parameters\n        deterministic=True,\n        seed=43,\n\n        # Data & model parameters\n        data=\"/kaggle/working/dataset/coco.yaml\", \n        save=True,\n        save_period=5,\n        pretrained=True,\n        imgsz=512,\n\n        # Training parameters\n        epochs=20,\n        batch=4,\n        workers=8,\n        val=True,\n        device=0,\n\n        # Optimization parameters\n        lr0=0.018,\n        patience=3,\n        optimizer=\"SGD\",\n        momentum=0.947,\n        weight_decay=0.0005,\n        close_mosaic=3,\n    )","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.75708Z","iopub.status.idle":"2023-07-21T07:50:34.757389Z","shell.execute_reply.started":"2023-07-21T07:50:34.757247Z","shell.execute_reply":"2023-07-21T07:50:34.757261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if __name__ == '__main__':\n#     main()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T07:50:34.758551Z","iopub.status.idle":"2023-07-21T07:50:34.758895Z","shell.execute_reply.started":"2023-07-21T07:50:34.758731Z","shell.execute_reply":"2023-07-21T07:50:34.758745Z"},"trusted":true},"execution_count":null,"outputs":[]}]}