{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Experiments for HuBMAP \nconducted by [Sayan Biswas](https://www.kaggle.com/sayanbiswas023) and [Sohith Bandari](https://www.kaggle.com/sohithbandari)\n\nNote: This notebook only lists the experiments meant to be run separately.....for enough epochs and enough computational power available","metadata":{"editable":false}},{"cell_type":"code","source":"## START\n!pip install ultralytics -q\n!pip install pycocotools -q","metadata":{"editable":false},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport random\nimport shutil\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom PIL import Image, ImageDraw \nfrom shapely.geometry import Polygon\n\nfrom ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:52:46.490896Z","iopub.execute_input":"2023-09-11T11:52:46.491449Z","iopub.status.idle":"2023-09-11T11:52:52.976602Z","shell.execute_reply.started":"2023-09-11T11:52:46.491408Z","shell.execute_reply":"2023-09-11T11:52:52.975529Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir=\"/kaggle/input/hubmap-hacking-the-human-vasculature\"\n! cd{root_dir}","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:52:52.978024Z","iopub.execute_input":"2023-09-11T11:52:52.978549Z","iopub.status.idle":"2023-09-11T11:52:54.015870Z","shell.execute_reply.started":"2023-09-11T11:52:52.978513Z","shell.execute_reply":"2023-09-11T11:52:54.014542Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\napi_key = user_secrets.get_secret(\"wandb\")\n\nimport wandb\nwandb.login(key=api_key)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:52:54.019818Z","iopub.execute_input":"2023-09-11T11:52:54.020559Z","iopub.status.idle":"2023-09-11T11:52:58.101253Z","shell.execute_reply.started":"2023-09-11T11:52:54.020526Z","shell.execute_reply":"2023-09-11T11:52:58.100083Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### FUNCTIONS TO LOAD,DISPLAY RAW TIFF FILES AND ANNOTATIONS ","metadata":{"editable":false}},{"cell_type":"code","source":"def tiff_loader(id: str,folder: str,display: bool=True):\n    file_path=os.path.join(\"/kaggle/input/hubmap-hacking-the-human-vasculature\",folder,(id+'.tif'))\n#     print(file_path)\n    image = Image.open(file_path)\n    if(display==True):\n        tiff_displayer(image)\n    return image\n        \n    \ndef tiff_displayer(image):\n    if image is not None:\n        plt.imshow(image)\n        plt.axis('off')  \n        plt.show()\n    else:\n        print('No image found to be displayed')\n\ndef make_outlines(image,polygon_vertices):\n    polygon = Polygon(polygon_vertices)\n    draw = ImageDraw.Draw(image)\n    fill_color = (0, 255, 0)  # Red color\n    outline_color = (255, 0, 0)  # Black outline color\n    draw.polygon(polygon.exterior.coords,outline=outline_color)\n\n    tiff_displayer(image)\n    \n    return image\n\ndef tiff_image_make_outlines(id,folder,polygon_vertices):\n    image=tiff_loader(id,folder,False)\n    polygon = Polygon(polygon_vertices)\n    draw = ImageDraw.Draw(image)\n    fill_color = (0, 255, 0)  # Red color\n    outline_color = (255, 0, 0)  # Black outline color\n    draw.polygon(polygon.exterior.coords,outline=outline_color)\n\n    tiff_displayer(image)\n    \n    return image\n\ndef tiff_labeler(id,folder,annot_dict):\n    color_dict={\n        'glomerulus':(255,0,0), #red\n        'blood_vessel':(0,255,0), #green\n        'unsure':(0,0,255) #blue\n    }\n    objects=annot_dict[id]\n    count_objects= len (objects)\n    \n    image=tiff_loader(id,folder,False)\n    \n    for obj in objects:\n        object_type=obj['type']\n        onject_polygon_vertices=obj['coordinates'][0]\n        polygon = Polygon(onject_polygon_vertices)\n        draw = ImageDraw.Draw(image)\n        \n        outline_color = color_dict[object_type]\n        draw.polygon(polygon.exterior.coords,outline=outline_color,width=3)\n\n    tiff_displayer(image)\n    \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:52:58.104068Z","iopub.execute_input":"2023-09-11T11:52:58.104730Z","iopub.status.idle":"2023-09-11T11:52:58.119534Z","shell.execute_reply.started":"2023-09-11T11:52:58.104698Z","shell.execute_reply":"2023-09-11T11:52:58.118471Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CREATES `annot_dict` TO STORE IMAGE_ID vs ANNOTATION LIST","metadata":{"editable":false}},{"cell_type":"code","source":"annotations=[]\nwith open(os.path.join(root_dir,'polygons.jsonl'), 'r') as f:\n    for line in tqdm(f):\n        annotations.append(json.loads(line))\nprint(len(annotations))\nprint(annotations[0].keys())\n\nannot_dict={} ## creates a list of annotation entries for each id: annptation entrie is a list of dict: dict.keys= ['type','coordinates']\nfor anot in tqdm(annotations):\n    annot_dict[anot['id']]=anot['annotations']","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:52:58.121119Z","iopub.execute_input":"2023-09-11T11:52:58.121578Z","iopub.status.idle":"2023-09-11T11:53:02.477393Z","shell.execute_reply.started":"2023-09-11T11:52:58.121542Z","shell.execute_reply":"2023-09-11T11:53:02.476407Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### USAGE EXAMPLE","metadata":{"editable":false}},{"cell_type":"code","source":"## USAGE: SINGLE FILE\n# tiff_labeler('0006ff2aa7cd','train',annot_dict)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:02.479154Z","iopub.execute_input":"2023-09-11T11:53:02.479939Z","iopub.status.idle":"2023-09-11T11:53:02.485166Z","shell.execute_reply.started":"2023-09-11T11:53:02.479881Z","shell.execute_reply":"2023-09-11T11:53:02.483984Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for _ in range(5):\n    idx = random.choice(list(annot_dict.keys()))\n    print(idx)\n    tiff_labeler(idx,'train',annot_dict)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:02.486579Z","iopub.execute_input":"2023-09-11T11:53:02.487061Z","iopub.status.idle":"2023-09-11T11:53:04.859822Z","shell.execute_reply.started":"2023-09-11T11:53:02.487027Z","shell.execute_reply":"2023-09-11T11:53:04.858822Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EXLORING METADATA","metadata":{"editable":false}},{"cell_type":"code","source":"tile=pd.read_csv(os.path.join(root_dir,'tile_meta.csv'))\nwsi=pd.read_csv(os.path.join(root_dir,'wsi_meta.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.861328Z","iopub.execute_input":"2023-09-11T11:53:04.862342Z","iopub.status.idle":"2023-09-11T11:53:04.899512Z","shell.execute_reply.started":"2023-09-11T11:53:04.862286Z","shell.execute_reply":"2023-09-11T11:53:04.898437Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.903736Z","iopub.execute_input":"2023-09-11T11:53:04.904359Z","iopub.status.idle":"2023-09-11T11:53:04.929384Z","shell.execute_reply.started":"2023-09-11T11:53:04.904328Z","shell.execute_reply":"2023-09-11T11:53:04.928413Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.931731Z","iopub.execute_input":"2023-09-11T11:53:04.932774Z","iopub.status.idle":"2023-09-11T11:53:04.949613Z","shell.execute_reply.started":"2023-09-11T11:53:04.932735Z","shell.execute_reply":"2023-09-11T11:53:04.948466Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CREATE DIRECTORIES","metadata":{"editable":false}},{"cell_type":"code","source":"# Creating Directories\n\nparent_dirpath = \"/kaggle/working/yolov8\"\n\nos.mkdir(parent_dirpath)\nos.mkdir(\"/kaggle/working/temp_images\")\nos.mkdir(\"/kaggle/working/temp_labels\")\n\nos.mkdir(os.path.join(parent_dirpath, \"train\"))\nos.mkdir(os.path.join(parent_dirpath, \"train\", \"images\"))\nos.mkdir(os.path.join(parent_dirpath, \"train\", \"labels\"))\n\nos.mkdir(os.path.join(parent_dirpath, \"val\"))\nos.mkdir(os.path.join(parent_dirpath, \"val\", \"images\"))\nos.mkdir(os.path.join(parent_dirpath, \"val\", \"labels\"))","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.950914Z","iopub.execute_input":"2023-09-11T11:53:04.951317Z","iopub.status.idle":"2023-09-11T11:53:04.961623Z","shell.execute_reply.started":"2023-09-11T11:53:04.951282Z","shell.execute_reply":"2023-09-11T11:53:04.960213Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tiff_to_jpg(file_name):\n    \n    tiff_image_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\" + str(file_name) + \".tif\"\n    tiff_image = Image.open(tiff_image_path)\n    destination_path = \"/kaggle/working/temp_images/\" + file_name + \".jpg\"\n    tiff_image.save(destination_path, 'JPEG')\n    \n    return 0\n\ndef vertices_to_txt(file_id, annotations, list_of_vertices):\n    \n    file_contents = []\n\n    for i in range(len(annotations)):\n\n        yolo_format = []\n        flag = 1\n\n        if annotations[i]['type'] == 'glomerulus':\n            yolo_format.append(str(1))\n            flag = 1\n        elif annotations[i]['type'] == 'blood_vessel':\n            yolo_format.append(str(0))\n            flag = 1\n        else:\n            flag = 0\n\n\n        if (flag):\n\n            list_of_vertices = annotations[i]['coordinates'][0]\n            for vertex in list_of_vertices:\n                yolo_format.append(str(vertex[0]/512))\n                yolo_format.append(str(vertex[1]/512))\n\n        yolo_format = \" \".join(yolo_format)\n\n        file_contents.append(yolo_format)\n\n    file_name = \"/kaggle/working/temp_labels/\" + str(file_id) + \".txt\"\n\n    with open(file_name, \"w\") as file:\n        if (len(file_contents) == 0):\n            pass\n        elif (len(file_contents) == 1):\n            file.write(str(file_contents[-1]))\n        else:\n            for k in range(len(file_contents)-1):\n                file.write(str(file_contents[k]) + \"\\n\")\n\n            file.write(str(file_contents[-1]))\n            \n    return 0","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.963275Z","iopub.execute_input":"2023-09-11T11:53:04.964038Z","iopub.status.idle":"2023-09-11T11:53:04.978584Z","shell.execute_reply.started":"2023-09-11T11:53:04.964000Z","shell.execute_reply":"2023-09-11T11:53:04.976986Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filepath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\n\nall_images = os.listdir(train_filepath)\nprint(\"No. of images:\", len(all_images))\n\njson_filepath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\nfile_ids = []\n\nwith open(json_filepath, 'r') as file:\n    \n    for line in file:\n        data = json.loads(line)\n        file_id = data['id']\n        annotations = data['annotations']\n        list_of_vertices = annotations[0]['coordinates'][0]\n        tiff_to_jpg(file_id)\n        vertices_to_txt(file_id, annotations, list_of_vertices)\n        file_ids.append(file_id)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:04.980337Z","iopub.execute_input":"2023-09-11T11:53:04.980873Z","iopub.status.idle":"2023-09-11T11:53:52.539951Z","shell.execute_reply.started":"2023-09-11T11:53:04.980748Z","shell.execute_reply":"2023-09-11T11:53:52.538899Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.shuffle(file_ids)\n\n# Train directory\nfor i in range(0, int(0.8*len(file_ids))):\n    \n    old_path_img = \"/kaggle/working/temp_images/\" + str(file_ids[i]) + \".jpg\"\n    new_path_img = \"/kaggle/working/yolov8/train/images/\" + str(file_ids[i]) + \".jpg\"\n    shutil.copy(old_path_img, new_path_img)\n    \n    old_path_txt = \"/kaggle/working/temp_labels/\" + str(file_ids[i]) + \".txt\"\n    new_path_txt = \"/kaggle/working/yolov8/train/labels/\" + str(file_ids[i]) + \".txt\"\n    shutil.copy(old_path_txt, new_path_txt)\n\n# Test directory\nfor i in range(int(0.8*len(file_ids)), len(file_ids)):\n    \n    old_path_img = \"/kaggle/working/temp_images/\" + str(file_ids[i]) + \".jpg\"\n    new_path_img = \"/kaggle/working/yolov8/val/images/\" + str(file_ids[i]) + \".jpg\"\n    shutil.copy(old_path_img, new_path_img)\n    \n    old_path_txt = \"/kaggle/working/temp_labels/\" + str(file_ids[i]) + \".txt\"\n    new_path_txt = \"/kaggle/working/yolov8/val/labels/\" + str(file_ids[i]) + \".txt\"\n    shutil.copy(old_path_txt, new_path_txt)\n    \nshutil.rmtree(\"/kaggle/working/temp_images\")\nshutil.rmtree(\"/kaggle/working/temp_labels\")","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:52.541727Z","iopub.execute_input":"2023-09-11T11:53:52.542132Z","iopub.status.idle":"2023-09-11T11:53:53.167432Z","shell.execute_reply.started":"2023-09-11T11:53:52.542094Z","shell.execute_reply":"2023-09-11T11:53:53.166395Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating custom_config.yaml file\nwith open(\"/kaggle/working/custom_config.yaml\", \"w\") as file:\n    file.write(\"path: /kaggle/working/yolov8\" + \"\\n\")\n    file.write(\"train: train/images\" + \"\\n\")\n    file.write(\"val: val/images\" + \"\\n\")\n    file.write(\"nc: 2\" + \"\\n\")\n    file.write(\"names: ['blood_vessel','glomerulus']\")","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:53.169046Z","iopub.execute_input":"2023-09-11T11:53:53.169433Z","iopub.status.idle":"2023-09-11T11:53:53.176526Z","shell.execute_reply.started":"2023-09-11T11:53:53.169391Z","shell.execute_reply":"2023-09-11T11:53:53.175414Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO('yolov8n-seg.pt')\n\nresults = model.train(data='/kaggle/working/custom_config.yaml',\n                      epochs=10,\n                      imgsz=512,\n                      # for augmentation\n                      degrees=90,\n                      translate=0.1,\n                      scale=0.5,\n                      flipud=0.5,\n                      fliplr=0.5)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T11:53:53.178347Z","iopub.execute_input":"2023-09-11T11:53:53.179222Z","iopub.status.idle":"2023-09-11T12:04:01.651794Z","shell.execute_reply.started":"2023-09-11T11:53:53.179186Z","shell.execute_reply":"2023-09-11T12:04:01.650271Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PREDICTION: \n## RUN IN SEPARATE NOTEBOOK","metadata":{"editable":false}},{"cell_type":"markdown","source":"We were struggling to understand the submission process [Mersico's](https://www.kaggle.com/mersico) notebook to the rescue. \nPls Refer.\n[Mersico's Notebook](https://www.kaggle.com/code/mersico/submission-for-the-little-ones-with-yolo)","metadata":{"editable":false}},{"cell_type":"code","source":"for root, dirs, files in os.walk(\"/kaggle/working/runs/segment/train/weights\"):\n    \n    print(f\"Current Directory: {root}\")\n    \n    print(\"Subdirectories:\")\n    for directory in dirs:\n        print(os.path.join(root, directory))\n    print(\"Files:\")\n    for file in files:\n        print(os.path.join(root, file))","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:05:44.092900Z","iopub.execute_input":"2023-09-11T12:05:44.093370Z","iopub.status.idle":"2023-09-11T12:05:44.107501Z","shell.execute_reply.started":"2023-09-11T12:05:44.093333Z","shell.execute_reply":"2023-09-11T12:05:44.106387Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import base64\nimport numpy as np\nimport torch\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\nimport pandas as pd\nimport torchvision.transforms as T","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:04:01.654197Z","iopub.execute_input":"2023-09-11T12:04:01.654826Z","iopub.status.idle":"2023-09-11T12:04:01.679762Z","shell.execute_reply.started":"2023-09-11T12:04:01.654788Z","shell.execute_reply":"2023-09-11T12:04:01.678693Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EncodeBinaryMask:\n    @staticmethod\n    def __checking_mask(mask: np.ndarray) -> np.ndarray:\n        if mask.dtype != np.bool:\n            raise ValueError(\n                \"expects a binary mask, received dtype == %s\" %\n                mask.dtype\n            )\n        return mask\n\n    @staticmethod\n    def __convert_mask(mask: np.ndarray):\n        mask_to_encode = mask.astype(np.uint8)\n        mask_to_encode = np.asfortranarray(mask_to_encode)\n        return mask_to_encode\n\n    @staticmethod\n    def __compress_encode(encoded_mask) -> t.Text:\n        binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n        base64_str = base64.b64encode(binary_str)\n        return base64_str\n\n    def __call__(self, mask: np.ndarray) -> t.Text:\n        mask = self.__checking_mask(mask)\n        mask_to_encode = self.__convert_mask(mask)\n        encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n        base64_str = self.__compress_encode(encoded_mask)\n        return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:06:04.601788Z","iopub.execute_input":"2023-09-11T12:06:04.602507Z","iopub.status.idle":"2023-09-11T12:06:04.617782Z","shell.execute_reply.started":"2023-09-11T12:06:04.602471Z","shell.execute_reply":"2023-09-11T12:06:04.615978Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Submission:\n    def __init__(self, dirpath: str, model: torch.nn.Module):\n        self.__eval_transforms = self.get_transforms()\n        self.__model = model\n        self.__encoder = EncodeBinaryMask()\n        self.__dirpath = dirpath\n        self.__filenames = os.listdir(dirpath)\n        self.height = 512\n        self.width = 512\n        \n        self.__submission_dict = {\n            \"id\": [],\n            \"height\": [],\n            \"width\": [],\n            \"prediction_string\": []\n        }\n        \n        self.submission = None\n    \n    @staticmethod\n    def get_transforms():\n        return T.Compose([\n            T.ToTensor(),\n            T.Resize(size=(512, 512)),\n            T.Normalize(mean=[0.485, 0.456, 0.406],\n                        std=[0.229, 0.224, 0.225])\n        ])\n\n    def __len__(self):\n        return len(self.__filenames)\n\n    def __get_columns(self) -> None:\n        for filename in self.__filenames:\n            path = self.__get_image_path(filename)\n            masks = self.__forward(path)\n            identifier, height, width, prediction_string = self.__get_cells(filename, masks)\n            self.__update_columns(identifier, height, width, prediction_string)\n\n    def __update_columns(self, identifier: str, height: int, width: int, prediction_string: str) -> None:\n        self.__submission_dict[\"id\"].append(identifier)\n        self.__submission_dict[\"height\"].append(height)\n        self.__submission_dict[\"width\"].append(width)\n        self.__submission_dict[\"prediction_string\"].append(prediction_string)\n\n    def __get_cells(self, filename: str, masks: list):\n        prediction_string = \"\"\n        prediction_string = self.__get_prediction_string(masks, prediction_string)\n        identifier = filename.split(\".\")[0]\n        return identifier, self.height, self.width, prediction_string\n\n    def __get_prediction_string(self, masks: list, prediction_string: str) -> str:\n        if masks:\n            for outputs in masks:\n                mask = outputs[\"mask\"]\n                mask = np.where(mask > 0.5, 1, 0).astype(np.bool)\n                base64_str = self.__encoder(mask)\n                confidence = outputs[\"confidence\"]\n                prediction_string += f\"0 {confidence} {base64_str.decode('utf-8')} \"\n        else:\n            return \"\"\n        return prediction_string\n\n    def __get_image_path(self, filename: str) -> str:\n        return os.path.join(\n            self.__dirpath, filename\n        )\n\n    def __get_image(self, path: str) -> torch.Tensor:\n        image = Image.open(path)\n        image = np.asarray(image)\n        image = self.__eval_transforms(image)\n        return image\n\n    def __forward(self, image: torch.tensor) -> list:\n        masks = self.__model(image) \n        return masks \n\n    def submit(self) -> None:\n        if not self.submission:\n            self.__get_columns()\n            self.submission = pd.DataFrame(self.__submission_dict)\n            self.submission = self.submission.set_index('id')\n            self.submission.to_csv(\"submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:06:06.267985Z","iopub.execute_input":"2023-09-11T12:06:06.268360Z","iopub.status.idle":"2023-09-11T12:06:06.291468Z","shell.execute_reply.started":"2023-09-11T12:06:06.268330Z","shell.execute_reply":"2023-09-11T12:06:06.289823Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BestYolo:\n    def __init__(self, conf: float = 0.05):\n        self.model_path = \"/kaggle/working/runs/segment/train/weights/best.pt\"\n        self.model = self.get_model()\n        self.conf = conf\n    \n    def get_model(self) -> YOLO:\n        return YOLO(self.model_path)\n    \n    def __call__(self, source) -> list[dict, ...]:\n        sublist = []\n        result = self.model(source)[0]\n        if result.masks:\n            for i in range(len(result.masks.data)):\n                conf = round(float(result.boxes.conf[i]), 2)\n                mask = np.expand_dims(result.masks.data[i].cpu().numpy(), axis=0).transpose(1,2,0)\n            \n                if int(result.boxes.cls[i]) == 0 and conf >= self.conf:\n                    sublist.append({\"mask\": mask, \"confidence\": conf})\n                else:\n                    continue\n            return sublist\n        else:\n            return None","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:06:07.870569Z","iopub.execute_input":"2023-09-11T12:06:07.870960Z","iopub.status.idle":"2023-09-11T12:06:07.882222Z","shell.execute_reply.started":"2023-09-11T12:06:07.870926Z","shell.execute_reply":"2023-09-11T12:06:07.880962Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"__TEST_PATH = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\nmodel = BestYolo()\nsub = Submission(dirpath=__TEST_PATH, model=model)\nsub.submit()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:06:17.923030Z","iopub.execute_input":"2023-09-11T12:06:17.923406Z","iopub.status.idle":"2023-09-11T12:06:18.330938Z","shell.execute_reply.started":"2023-09-11T12:06:17.923376Z","shell.execute_reply":"2023-09-11T12:06:18.329692Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:06:44.242931Z","iopub.execute_input":"2023-09-11T12:06:44.243329Z","iopub.status.idle":"2023-09-11T12:06:44.264926Z","shell.execute_reply.started":"2023-09-11T12:06:44.243296Z","shell.execute_reply":"2023-09-11T12:06:44.263851Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:08:35.020427Z","iopub.execute_input":"2023-09-11T12:08:35.020810Z","iopub.status.idle":"2023-09-11T12:08:35.033411Z","shell.execute_reply.started":"2023-09-11T12:08:35.020778Z","shell.execute_reply":"2023-09-11T12:08:35.032459Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EXPERIMENT PART 2: \n## SEMI SUPERVISED LEARNING \n[REFER NOTEBOOK](https://www.kaggle.com/code/sohithbandari/hubmap-yolov8-semi-supervised)","metadata":{"editable":false}},{"cell_type":"code","source":"## Note experiment will take long to run...run in separate notebook\nmodel = YOLO('yolov8x-seg.pt')\n\nresults = model.train(data='/kaggle/working/custom_config.yaml',\n                      epochs=100,\n                      imgsz=512,\n#                       device=[0, 1],\n                      optimizer='Adam',\n                      seed=42,\n                      close_mosaic=0,\n                      mask_ratio=1,\n                      val=False,\n                      # for augmentation\n                      degrees=90,\n                      translate=0.1,\n                      scale=0.5,\n                      flipud=0.5,\n                      fliplr=0.5)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:25:58.769784Z","iopub.execute_input":"2023-09-11T12:25:58.770723Z","iopub.status.idle":"2023-09-11T12:25:58.777836Z","shell.execute_reply.started":"2023-09-11T12:25:58.770674Z","shell.execute_reply":"2023-09-11T12:25:58.776717Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.predict(\"/kaggle/working/yolov8/test/images/3fa3016e2bc5.jpg\")\ncount = 0\n\nfor result in results:\n    boxes = result.boxes.conf\n    classes = result.boxes.cls\n    masks = result.masks.xyn\n    count += 1\n    \nprint(count)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T12:25:59.207500Z","iopub.execute_input":"2023-09-11T12:25:59.207872Z","iopub.status.idle":"2023-09-11T12:25:59.216788Z","shell.execute_reply.started":"2023-09-11T12:25:59.207843Z","shell.execute_reply":"2023-09-11T12:25:59.213263Z"},"editable":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_added = 0\n\nfor file_id in tqdm(unlabelled_images):\n    \n    tiff_image_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\" + str(file_id) + \".tif\"\n    tiff_image = Image.open(tiff_image_path)\n    destination_path = \"/kaggle/working/temp_image.jpg\"\n    tiff_image.save(destination_path, 'JPEG')\n\n    results = model.predict(destination_path, verbose=False)\n    \n    flag = 1\n    file_contents = []\n    \n    for result in results:\n        boxes = result.boxes.conf\n        if len(boxes) != 0:\n            classes = result.boxes.cls\n            masks = result.masks.xyn\n        else:\n            flag = 0\n\n    if (flag):\n        for i in range(len(boxes)):\n            if boxes[i] < 0.4:\n                flag=0\n                break\n\n    if(flag):\n        des_img_filepath = os.path.join(\"/kaggle/working/yolov8/train/images/\" + str(file_id) + \".jpg\")\n        shutil.copy(destination_path, des_img_filepath)\n\n        for i in range(len(boxes)):\n\n            yolo_format = []\n\n            if classes[i] == 1:\n                yolo_format.append(str(1))\n            else:\n                yolo_format.append(str(0))\n\n            list_of_vertices = masks[i]\n            for vertex in list_of_vertices:\n                yolo_format.append(str(vertex[0]))\n                yolo_format.append(str(vertex[1]))\n\n            yolo_format = \" \".join(yolo_format)\n\n            file_contents.append(yolo_format)\n\n        file_name = os.path.join(\"/kaggle/working/yolov8/train/labels/\" + str(file_id) + \".txt\")\n\n        with open(file_name, \"w\") as file:\n            if (len(file_contents) == 1):\n                file.write(str(file_contents[-1]))\n            else:\n                for k in range(len(file_contents)-1):\n                    file.write(str(file_contents[k]) + \"\\n\")\n\n                file.write(str(file_contents[-1]))\n\n        images_added += 1\n    \n    flag = 1\n\nprint(\"Images added to training set:\", images_added)","metadata":{"editable":false},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(list(os.listdir(\"/kaggle/working/yolov8/train/labels\")))","metadata":{"editable":false},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO('yolov8x-seg.pt')\n\nresults = model.train(data='/kaggle/working/custom_config.yaml',\n                      epochs=100,\n                      imgsz=512,\n#                       device=[0, 1],\n                      optimizer='Adam',\n                      seed=42,\n                      close_mosaic=0,\n                      mask_ratio=1,\n                      val=False,\n                      # for augmentation\n                      degrees=90,\n                      translate=0.1,\n                      scale=0.5,\n                      flipud=0.5,\n                      fliplr=0.5)","metadata":{"editable":false},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EXPERIMENT PART 3: \n## SEMI SUPERVISED LEARNING: PUT YOLO SSL TRAIN IN LOOP\n\n[Refer Notebook](https://www.kaggle.com/sayanbiswas023/hubmap-yolov8-semi-supervised-version-2)","metadata":{"editable":false}},{"cell_type":"markdown","source":"Experiment with smaller YOLO model and introducing semi supervision after training for every super_epoch ","metadata":{"editable":false}},{"cell_type":"code","source":"## Predict on unlabbeled files and add them to train\n\ndef predict_and_add_images(ssl_epoch,model):\n    \n    images_added = 0\n    \n    for file_id in unlabelled_images:\n\n        tiff_image_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\" + str(file_id) + \".tif\"\n        tiff_image = Image.open(tiff_image_path)\n        destination_path = \"/kaggle/working/temp_image.jpg\"\n        tiff_image.save(destination_path, 'JPEG')\n\n        results = model.predict(destination_path, verbose=False)\n\n        flag = 1\n        file_contents = []\n\n        for result in results:\n            boxes = result.boxes.conf\n            if len(boxes) != 0:\n                classes = result.boxes.cls\n                masks = result.masks.xyn\n            else:\n                flag = 0\n\n        if (flag):\n            for i in range(len(boxes)):\n                if boxes[i] < 0.2:    ## Set threshold of confidence\n                    flag=0\n                    break\n\n        if(flag):\n            des_img_filepath = os.path.join(\"/kaggle/working/yolov8/train/images/\" + str(file_id) + \".jpg\")\n            shutil.copy(destination_path, des_img_filepath)\n            unlabelled_images.remove(file_id)\n\n            for i in range(len(boxes)):\n\n                yolo_format = []\n\n                if classes[i] == 1:\n                    yolo_format.append(str(1))\n                else:\n                    yolo_format.append(str(0))\n\n                list_of_vertices = masks[i]\n                for vertex in list_of_vertices:\n                    yolo_format.append(str(vertex[0]))\n                    yolo_format.append(str(vertex[1]))\n\n                yolo_format = \" \".join(yolo_format)\n\n                file_contents.append(yolo_format)\n\n            file_name = os.path.join(\"/kaggle/working/yolov8/train/labels/\" + str(file_id) + \".txt\")\n\n            with open(file_name, \"w\") as file:\n                if (len(file_contents) == 1):\n                    file.write(str(file_contents[-1]))\n                else:\n                    for k in range(len(file_contents)-1):\n                        file.write(str(file_contents[k]) + \"\\n\")\n\n                    file.write(str(file_contents[-1]))\n\n            images_added += 1\n\n        flag = 1\n    \n    print(\"#########################################################################\")\n    print(f\" For SSL Epoch {ssl_epoch} Images added to training set: {images_added}\" )\n    print(f\" Number of images left in unlabbeld images {len(unlabelled_images)}\" )\n    print(\"#########################################################################\")","metadata":{"editable":false},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssl_epochs = 5\n\nmodel = YOLO('yolov8n-seg.pt')\n\nfor ep in tqdm(range(ssl_epochs)):\n    print(f\"Starting Train for {ep+1} super_epoch\")\n\n    results = model.train(data='/kaggle/working/custom_config.yaml',\n                          epochs=1,\n                          imgsz=512,\n#                           device=[0, 1],\n                          optimizer='Adam',\n                          seed=42,\n                          close_mosaic=0,\n                          mask_ratio=1,\n                          val=False,\n                          verbose=False,\n                          # for augmentation\n                          degrees=90,\n                          translate=0.1,\n                          scale=0.5,\n                          flipud=0.5,\n                          fliplr=0.5)\n    \n    predict_and_add_images(ep,model)\n    ","metadata":{"editable":false},"execution_count":null,"outputs":[]}]}