{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics==8.0.176 -q\n!pip install pycocotools -q","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-22T15:20:10.555890Z","iopub.execute_input":"2023-09-22T15:20:10.556372Z","iopub.status.idle":"2023-09-22T15:20:37.534496Z","shell.execute_reply.started":"2023-09-22T15:20:10.556329Z","shell.execute_reply":"2023-09-22T15:20:37.533286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport random\nimport shutil\nfrom tqdm import tqdm\n\nfrom PIL import Image\nfrom ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:37.537815Z","iopub.execute_input":"2023-09-22T15:20:37.538531Z","iopub.status.idle":"2023-09-22T15:20:41.653907Z","shell.execute_reply.started":"2023-09-22T15:20:37.538495Z","shell.execute_reply":"2023-09-22T15:20:41.652955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\napi_key = user_secrets.get_secret(\"wandb\")\n\nimport wandb\nwandb.login(key=api_key)","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:41.655433Z","iopub.execute_input":"2023-09-22T15:20:41.656091Z","iopub.status.idle":"2023-09-22T15:20:44.893640Z","shell.execute_reply.started":"2023-09-22T15:20:41.656052Z","shell.execute_reply":"2023-09-22T15:20:44.892648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir=\"/kaggle/input/hubmap-hacking-the-human-vasculature\"\n! cd{root_dir}","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:44.896397Z","iopub.execute_input":"2023-09-22T15:20:44.896881Z","iopub.status.idle":"2023-09-22T15:20:45.854866Z","shell.execute_reply.started":"2023-09-22T15:20:44.896853Z","shell.execute_reply":"2023-09-22T15:20:45.853348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations=[]\nwith open(os.path.join(root_dir,'polygons.jsonl'), 'r') as f:\n    for line in tqdm(f):\n        annotations.append(json.loads(line))\nprint(len(annotations))\nprint(annotations[0].keys())\n\nannot_dict={} ## creates a list of annotation entries for each id: annptation entrie is a list of dict: dict.keys= ['type','coordinates']\nfor anot in tqdm(annotations):\n    annot_dict[anot['id']]=anot['annotations']","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:45.856519Z","iopub.execute_input":"2023-09-22T15:20:45.857938Z","iopub.status.idle":"2023-09-22T15:20:49.934053Z","shell.execute_reply.started":"2023-09-22T15:20:45.857898Z","shell.execute_reply":"2023-09-22T15:20:49.933076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Directories\n\nparent_dirpath = \"/kaggle/working/yolov8\"\n\nos.mkdir(parent_dirpath)\nos.mkdir(\"/kaggle/working/temp_images\")\nos.mkdir(\"/kaggle/working/temp_labels\")\n\nos.mkdir(os.path.join(parent_dirpath, \"train\"))\nos.mkdir(os.path.join(parent_dirpath, \"train\", \"images\"))\nos.mkdir(os.path.join(parent_dirpath, \"train\", \"labels\"))\n\nos.mkdir(os.path.join(parent_dirpath, \"test\"))\nos.mkdir(os.path.join(parent_dirpath, \"test\", \"images\"))\nos.mkdir(os.path.join(parent_dirpath, \"test\", \"labels\"))","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:49.935586Z","iopub.execute_input":"2023-09-22T15:20:49.937474Z","iopub.status.idle":"2023-09-22T15:20:49.945108Z","shell.execute_reply.started":"2023-09-22T15:20:49.937438Z","shell.execute_reply":"2023-09-22T15:20:49.944208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tiff_to_jpg(file_name):\n    \n    tiff_image_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\" + str(file_name) + \".tif\"\n    tiff_image = Image.open(tiff_image_path)\n    destination_path = \"/kaggle/working/temp_images/\" + file_name + \".jpg\"\n    tiff_image.save(destination_path, 'JPEG')\n    \n    return 0\n\ndef vertices_to_txt(file_id, annotations, list_of_vertices):\n    \n    file_contents = []\n\n    for i in range(len(annotations)):\n\n        yolo_format = []\n        flag = 1\n\n        if annotations[i]['type'] == 'glomerulus':\n            yolo_format.append(str(1))\n            flag = 1\n        elif annotations[i]['type'] == 'blood_vessel':\n            yolo_format.append(str(0))\n            flag = 1\n        else:\n            flag = 0\n\n\n        if (flag):\n\n            list_of_vertices = annotations[i]['coordinates'][0]\n            for vertex in list_of_vertices:\n                yolo_format.append(str(vertex[0]/512))\n                yolo_format.append(str(vertex[1]/512))\n\n        yolo_format = \" \".join(yolo_format)\n\n        file_contents.append(yolo_format)\n\n    file_name = \"/kaggle/working/temp_labels/\" + str(file_id) + \".txt\"\n\n    with open(file_name, \"w\") as file:\n        if (len(file_contents) == 0):\n            pass\n        elif (len(file_contents) == 1):\n            file.write(str(file_contents[-1]))\n        else:\n            for k in range(len(file_contents)-1):\n                file.write(str(file_contents[k]) + \"\\n\")\n\n            file.write(str(file_contents[-1]))\n            \n    return 0","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:49.946428Z","iopub.execute_input":"2023-09-22T15:20:49.947210Z","iopub.status.idle":"2023-09-22T15:20:49.960490Z","shell.execute_reply.started":"2023-09-22T15:20:49.947174Z","shell.execute_reply":"2023-09-22T15:20:49.959514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dest=\"/kaggle/working/unlabelled_images\"\n# for root, dirs, files in os.walk(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"):\n    \n#     for file in files:\n        \n#         fileid=file.split('.')[0]\n#         if fileid not in annot_dict.keys():\n#             tiff_to_jpg_with_dest(fileid,dest)\n\n# print(len(os.listdir(\"/kaggle/working/train_images\")))\n# print(len(os.listdir(\"/kaggle/working/train_labels\")))\n# print(len(os.listdir(\"/kaggle/working/unlabelled_images\")))","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:49.961907Z","iopub.execute_input":"2023-09-22T15:20:49.962293Z","iopub.status.idle":"2023-09-22T15:20:49.975941Z","shell.execute_reply.started":"2023-09-22T15:20:49.962261Z","shell.execute_reply":"2023-09-22T15:20:49.975001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_filepath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\nfile_ids = []\n\nwith open(json_filepath, 'r') as file:\n    \n    for line in file:\n        data = json.loads(line)\n        file_id = data['id']\n        annotations = data['annotations']\n        list_of_vertices = annotations[0]['coordinates'][0]\n        tiff_to_jpg(file_id)\n        vertices_to_txt(file_id, annotations, list_of_vertices)\n        file_ids.append(file_id)","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:20:49.977360Z","iopub.execute_input":"2023-09-22T15:20:49.977979Z","iopub.status.idle":"2023-09-22T15:21:37.140930Z","shell.execute_reply.started":"2023-09-22T15:20:49.977945Z","shell.execute_reply":"2023-09-22T15:21:37.139939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filepath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\n\nall_images = os.listdir(train_filepath)\nprint(\"No. of images:\", len(all_images))\n\nall_file_ids = []\n\nfor file_name in all_images:\n    file_name = file_name.split('.')\n    all_file_ids.append(file_name[0])\n\nunlabelled_images = list(set(all_file_ids).difference(file_ids))","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:21:37.144261Z","iopub.execute_input":"2023-09-22T15:21:37.145036Z","iopub.status.idle":"2023-09-22T15:21:37.513984Z","shell.execute_reply.started":"2023-09-22T15:21:37.145004Z","shell.execute_reply":"2023-09-22T15:21:37.513054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.shuffle(file_ids)\n\n# Train directory\nfor i in range(0, int(0.8*len(file_ids))):\n    \n    old_path_img = \"/kaggle/working/temp_images/\" + str(file_ids[i]) + \".jpg\"\n    new_path_img = \"/kaggle/working/yolov8/train/images/\" + str(file_ids[i]) + \".jpg\"\n    shutil.copy(old_path_img, new_path_img)\n    \n    old_path_txt = \"/kaggle/working/temp_labels/\" + str(file_ids[i]) + \".txt\"\n    new_path_txt = \"/kaggle/working/yolov8/train/labels/\" + str(file_ids[i]) + \".txt\"\n    shutil.copy(old_path_txt, new_path_txt)\n\n# Test directory\nfor i in range(int(0.8*len(file_ids)), len(file_ids)):\n    \n    old_path_img = \"/kaggle/working/temp_images/\" + str(file_ids[i]) + \".jpg\"\n    new_path_img = \"/kaggle/working/yolov8/test/images/\" + str(file_ids[i]) + \".jpg\"\n    shutil.copy(old_path_img, new_path_img)\n    \n    old_path_txt = \"/kaggle/working/temp_labels/\" + str(file_ids[i]) + \".txt\"\n    new_path_txt = \"/kaggle/working/yolov8/test/labels/\" + str(file_ids[i]) + \".txt\"\n    shutil.copy(old_path_txt, new_path_txt)\n    \nshutil.rmtree(\"/kaggle/working/temp_images\")\nshutil.rmtree(\"/kaggle/working/temp_labels\")","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:21:37.515204Z","iopub.execute_input":"2023-09-22T15:21:37.515569Z","iopub.status.idle":"2023-09-22T15:21:38.087626Z","shell.execute_reply.started":"2023-09-22T15:21:37.515536Z","shell.execute_reply":"2023-09-22T15:21:38.086444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unlabelled_images)","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:21:38.089245Z","iopub.execute_input":"2023-09-22T15:21:38.089623Z","iopub.status.idle":"2023-09-22T15:21:38.096334Z","shell.execute_reply.started":"2023-09-22T15:21:38.089588Z","shell.execute_reply":"2023-09-22T15:21:38.095285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating custom_config.yaml file\nwith open(\"/kaggle/working/custom_config.yaml\", \"w\") as file:\n    file.write(\"path: /kaggle/working/yolov8\" + \"\\n\")\n    file.write(\"train: train/images\" + \"\\n\")\n    file.write(\"val: test/images\" + \"\\n\")\n    file.write(\"test: test/images\" + \"\\n\")\n    file.write(\"nc: 2\" + \"\\n\")\n    file.write(\"names: ['blood_vessel','glomerulus']\")","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:21:38.098148Z","iopub.execute_input":"2023-09-22T15:21:38.099016Z","iopub.status.idle":"2023-09-22T15:21:38.108178Z","shell.execute_reply.started":"2023-09-22T15:21:38.098910Z","shell.execute_reply":"2023-09-22T15:21:38.107148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Predict on unlabbeled files and add them to train\n\ndef predict_and_add_images(ssl_epoch,model):\n    \n    images_added = 0\n    \n    for file_id in unlabelled_images:\n\n        tiff_image_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\" + str(file_id) + \".tif\"\n        tiff_image = Image.open(tiff_image_path)\n        destination_path = \"/kaggle/working/temp_image.jpg\"\n        tiff_image.save(destination_path, 'JPEG')\n\n        results = model.predict(destination_path, verbose=False)\n\n        flag = 1\n        file_contents = []\n\n        for result in results:\n            boxes = result.boxes.conf\n            if len(boxes) != 0:\n                classes = result.boxes.cls\n                masks = result.masks.xyn\n            else:\n                flag = 0\n\n        if (flag):\n            for i in range(len(boxes)):\n                if boxes[i] < 0.2:    ## Set threshold of confidence\n                    flag=0\n                    break\n\n        if(flag):\n            des_img_filepath = os.path.join(\"/kaggle/working/yolov8/train/images/\" + str(file_id) + \".jpg\")\n            shutil.copy(destination_path, des_img_filepath)\n            unlabelled_images.remove(file_id)\n\n            for i in range(len(boxes)):\n\n                yolo_format = []\n\n                if classes[i] == 1:\n                    yolo_format.append(str(1))\n                else:\n                    yolo_format.append(str(0))\n\n                list_of_vertices = masks[i]\n                for vertex in list_of_vertices:\n                    yolo_format.append(str(vertex[0]))\n                    yolo_format.append(str(vertex[1]))\n\n                yolo_format = \" \".join(yolo_format)\n\n                file_contents.append(yolo_format)\n\n            file_name = os.path.join(\"/kaggle/working/yolov8/train/labels/\" + str(file_id) + \".txt\")\n\n            with open(file_name, \"w\") as file:\n                if (len(file_contents) == 1):\n                    file.write(str(file_contents[-1]))\n                else:\n                    for k in range(len(file_contents)-1):\n                        file.write(str(file_contents[k]) + \"\\n\")\n\n                    file.write(str(file_contents[-1]))\n\n            images_added += 1\n\n        flag = 1\n    \n    print(\"#########################################################################\")\n    print(f\" For SSL Epoch {ssl_epoch} Images added to training set: {images_added}\" )\n    print(f\" Number of images left in unlabbeld images {len(unlabelled_images)}\" )\n    print(\"#########################################################################\")","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:31:09.675540Z","iopub.execute_input":"2023-09-22T15:31:09.676302Z","iopub.status.idle":"2023-09-22T15:31:09.700014Z","shell.execute_reply.started":"2023-09-22T15:31:09.676255Z","shell.execute_reply":"2023-09-22T15:31:09.698965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssl_epochs = 5\n\nmodel = YOLO('yolov8n-seg.pt')\n\nfor ep in tqdm(range(ssl_epochs)):\n    print(f\"Starting Train for {ep+1} super_epoch\")\n\n    results = model.train(data='/kaggle/working/custom_config.yaml',\n                          epochs=1,\n                          imgsz=512,\n#                           device=[0, 1],\n                          optimizer='Adam',\n                          seed=42,\n                          close_mosaic=0,\n                          mask_ratio=1,\n                          val=False,\n                          verbose=False,\n                          # for augmentation\n                          degrees=90,\n                          translate=0.1,\n                          scale=0.5,\n                          flipud=0.5,\n                          fliplr=0.5)\n    \n    predict_and_add_images(ep,model)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-09-22T15:31:28.076583Z","iopub.execute_input":"2023-09-22T15:31:28.076960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}