{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob\nimport json\nimport os\nimport cv2\nimport yaml\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-10T10:50:51.954504Z","iopub.execute_input":"2024-01-10T10:50:51.954912Z","iopub.status.idle":"2024-01-10T10:50:51.959760Z","shell.execute_reply.started":"2024-01-10T10:50:51.954882Z","shell.execute_reply":"2024-01-10T10:50:51.958751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert tif labels to coco json format","metadata":{}},{"cell_type":"code","source":"MASK_EXT = 'tif'\nORIGINAL_EXT = 'tif'\nMASK_PATH = 'labels'\nIMG_PATH = 'images'","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:50:51.961340Z","iopub.execute_input":"2024-01-10T10:50:51.961623Z","iopub.status.idle":"2024-01-10T10:50:51.976817Z","shell.execute_reply.started":"2024-01-10T10:50:51.961599Z","shell.execute_reply":"2024-01-10T10:50:51.975912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_annotation_for_contour(contour, annotation_id: int, image_id):\n    bbox = cv2.boundingRect(contour)\n    area = cv2.contourArea(contour)\n    segmentation = contour.flatten().tolist()\n\n    annotation = {\n        \"iscrowd\": 0,\n        \"id\": annotation_id,\n        \"image_id\": image_id,\n        \"category_id\": 1,\n        \"bbox\": bbox,\n        \"area\": area,\n        \"segmentation\": [segmentation],\n    }\n\n    return annotation\n\n\ndef contours_from_mask_image(mask_image_open):\n    # Find contours in the mask image\n    gray = cv2.cvtColor(mask_image_open, cv2.COLOR_BGR2GRAY)\n    _, thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    contours = cv2.findContours(thresh, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)[0]\n    return contours\n\n\ndef images_annotations_info(path):\n    \"\"\"\n    Process the binary masks and generate images and annotations information.\n\n    :param path: Path to the directory containing images and binary masks\n    :return: Tuple containing images info, annotations info, and annotation count\n    \"\"\"\n    global image_id, annotation_id\n    annotations = []\n    images = []\n\n\n    for mask_image in glob.glob(os.path.join(path, MASK_PATH, f'*.{MASK_EXT}')):\n        original_file_name = f'{os.path.basename(mask_image).split(\".\")[0]}.{ORIGINAL_EXT}'\n        mask_image_open = cv2.imread(mask_image)\n\n        # Get image dimensions\n        height, width, _ = mask_image_open.shape\n\n        # Create or find existing image annotation\n        if original_file_name not in map(lambda img: img['file_name'], images):\n            image = {\n                \"id\": image_id + 1,\n                \"width\": width,\n                \"height\": height,\n                \"file_name\": original_file_name,\n            }\n            images.append(image)\n            image_id += 1\n        else:\n            image = [element for element in images if element['file_name'] == original_file_name][0]\n\n        contours = contours_from_mask_image(mask_image_open)\n\n        # Create annotation for each contour\n        for contour in contours:\n            annotation = create_annotation_for_contour(contour, annotation_id, image['id'])\n\n            # Add annotation if area is greater than zero\n            if annotation[\"area\"] > 0:\n                annotations.append(annotation)\n                annotation_id += 1\n\n    return images, annotations, annotation_id\n\n\ndef process_masks(mask_path, dest_json):\n    # Initialize the COCO JSON format with categories\n    coco_format = {\n        \"info\": {},\n        \"licenses\": [],\n        \"images\": [],\n        \"categories\": [{\"id\": 1, \"name\": 'Vessel', \"supercategory\": 'Vessel'}],\n        \"annotations\": [],\n    }\n\n    # Create images and annotations sections\n    coco_format[\"images\"], coco_format[\"annotations\"], annotation_cnt = images_annotations_info(mask_path)\n\n    # Save the COCO JSON to a file\n    with open(dest_json, \"w\") as outfile:\n        json.dump(coco_format, outfile, sort_keys=True, indent=4)\n\n    print(f\"Created {annotation_cnt} annotations for images in folder: {mask_path}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:50:52.056218Z","iopub.execute_input":"2024-01-10T10:50:52.056481Z","iopub.status.idle":"2024-01-10T10:50:52.071295Z","shell.execute_reply.started":"2024-01-10T10:50:52.056459Z","shell.execute_reply":"2024-01-10T10:50:52.070368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p small/train/images\n!mkdir -p small/val/images\n\n!mkdir -p small/train/labels\n!mkdir -p small/val/labels","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:50:52.072923Z","iopub.execute_input":"2024-01-10T10:50:52.073233Z","iopub.status.idle":"2024-01-10T10:50:55.995095Z","shell.execute_reply.started":"2024-01-10T10:50:52.073209Z","shell.execute_reply":"2024-01-10T10:50:55.993984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy images 0500.tif - 0599.tif\n!cp /kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/images/05**.tif /kaggle/working/small/train/images\n!cp /kaggle/input/blood-vessel-segmentation/train/kidney_3_sparse/images/05**.tif /kaggle/working/small/val/images\n","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:50:55.996738Z","iopub.execute_input":"2024-01-10T10:50:55.997143Z","iopub.status.idle":"2024-01-10T10:51:04.921006Z","shell.execute_reply.started":"2024-01-10T10:50:55.997090Z","shell.execute_reply":"2024-01-10T10:51:04.919703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy labels 0500.tif - 0599.tif\n!cp /kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/labels/05**.tif /kaggle/working/small/train/labels\n!cp /kaggle/input/blood-vessel-segmentation/train/kidney_3_dense/labels/05**.tif /kaggle/working/small/val/labels\n","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:04.922457Z","iopub.execute_input":"2024-01-10T10:51:04.922746Z","iopub.status.idle":"2024-01-10T10:51:09.746350Z","shell.execute_reply.started":"2024-01-10T10:51:04.922720Z","shell.execute_reply":"2024-01-10T10:51:09.744976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity check to see if we have 100 files in each dir\n!ls -l small/val/labels | grep \"^-\" | wc -l\n!ls -l small/val/images | grep \"^-\" | wc -l\n!ls -l small/train/labels | grep \"^-\" | wc -l\n!ls -l small/train/images | grep \"^-\" | wc -l\n","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:09.749703Z","iopub.execute_input":"2024-01-10T10:51:09.750012Z","iopub.status.idle":"2024-01-10T10:51:13.690250Z","shell.execute_reply.started":"2024-01-10T10:51:09.749985Z","shell.execute_reply":"2024-01-10T10:51:13.688926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global image_id, annotation_id\nimage_id = 0\nannotation_id = 0","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:13.692042Z","iopub.execute_input":"2024-01-10T10:51:13.692495Z","iopub.status.idle":"2024-01-10T10:51:13.698162Z","shell.execute_reply.started":"2024-01-10T10:51:13.692453Z","shell.execute_reply":"2024-01-10T10:51:13.697184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/working/small/train'\nval_path = '/kaggle/working/small/val'\n\ntrain_json_path = '/kaggle/working/small/train/train.json'\nval_json_path = '/kaggle/working/small/val/val.json'","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:13.699527Z","iopub.execute_input":"2024-01-10T10:51:13.699845Z","iopub.status.idle":"2024-01-10T10:51:13.710188Z","shell.execute_reply.started":"2024-01-10T10:51:13.699818Z","shell.execute_reply":"2024-01-10T10:51:13.709156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_masks(train_path, train_json_path)\nprocess_masks(val_path, val_json_path)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:13.711524Z","iopub.execute_input":"2024-01-10T10:51:13.711807Z","iopub.status.idle":"2024-01-10T10:51:18.263088Z","shell.execute_reply.started":"2024-01-10T10:51:13.711783Z","shell.execute_reply":"2024-01-10T10:51:18.262164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm /kaggle/working/small/train/labels/*.tif\n!rm /kaggle/working/small/val/labels/*.tif","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:18.264345Z","iopub.execute_input":"2024-01-10T10:51:18.264631Z","iopub.status.idle":"2024-01-10T10:51:20.257281Z","shell.execute_reply.started":"2024-01-10T10:51:18.264606Z","shell.execute_reply":"2024-01-10T10:51:20.256108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tail -n 100 /kaggle/working/small/val/val.json","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:20.258824Z","iopub.execute_input":"2024-01-10T10:51:20.259185Z","iopub.status.idle":"2024-01-10T10:51:21.248647Z","shell.execute_reply.started":"2024-01-10T10:51:20.259142Z","shell.execute_reply":"2024-01-10T10:51:21.247447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert annotations from coco json to yolo format using ultralytics `convert_coco`","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:21.250216Z","iopub.execute_input":"2024-01-10T10:51:21.250549Z","iopub.status.idle":"2024-01-10T10:51:37.167106Z","shell.execute_reply.started":"2024-01-10T10:51:21.250521Z","shell.execute_reply":"2024-01-10T10:51:37.165934Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics.data.converter import convert_coco","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:37.168814Z","iopub.execute_input":"2024-01-10T10:51:37.169445Z","iopub.status.idle":"2024-01-10T10:51:41.723948Z","shell.execute_reply.started":"2024-01-10T10:51:37.169405Z","shell.execute_reply":"2024-01-10T10:51:41.723127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert_coco(labels_dir='/kaggle/working/small/train', \n             use_segments=True,\n             cls91to80=False,)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:41.727201Z","iopub.execute_input":"2024-01-10T10:51:41.727641Z","iopub.status.idle":"2024-01-10T10:51:42.527157Z","shell.execute_reply.started":"2024-01-10T10:51:41.727612Z","shell.execute_reply":"2024-01-10T10:51:42.526307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp  /kaggle/working/coco_converted/labels/train/*.txt /kaggle/working/small/train/labels/\n!rm -r /kaggle/working/coco_converted","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:42.528275Z","iopub.execute_input":"2024-01-10T10:51:42.528572Z","iopub.status.idle":"2024-01-10T10:51:44.500042Z","shell.execute_reply.started":"2024-01-10T10:51:42.528547Z","shell.execute_reply":"2024-01-10T10:51:44.498709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert_coco(labels_dir='/kaggle/working/small/val', \n             use_segments=True,\n             cls91to80=False,)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:44.502001Z","iopub.execute_input":"2024-01-10T10:51:44.502395Z","iopub.status.idle":"2024-01-10T10:51:45.195973Z","shell.execute_reply.started":"2024-01-10T10:51:44.502364Z","shell.execute_reply":"2024-01-10T10:51:45.194961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp  /kaggle/working/coco_converted/labels/val/*.txt /kaggle/working/small/val/labels/\n!rm -r /kaggle/working/coco_converted","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:45.197187Z","iopub.execute_input":"2024-01-10T10:51:45.197457Z","iopub.status.idle":"2024-01-10T10:51:47.149034Z","shell.execute_reply.started":"2024-01-10T10:51:45.197432Z","shell.execute_reply":"2024-01-10T10:51:47.147873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Write yolo dataset YAML","metadata":{}},{"cell_type":"code","source":"names = {0: 'Vessel'}\n\n# Number of classes\nnc = len(names)\n\n# Create a dictionary with the required content\nyaml_data = {\n    'names': names,\n    'nc': nc,\n    'test': '',\n    'train': train_path,\n    'val': val_path\n}\n\n# Write the dictionary to a YAML file\nwith open('small/small_yolo.yaml', 'w') as file:\n    yaml.dump(yaml_data, file, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:47.150519Z","iopub.execute_input":"2024-01-10T10:51:47.150822Z","iopub.status.idle":"2024-01-10T10:51:47.158414Z","shell.execute_reply.started":"2024-01-10T10:51:47.150786Z","shell.execute_reply":"2024-01-10T10:51:47.157514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat small/small_yolo.yaml","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:47.159537Z","iopub.execute_input":"2024-01-10T10:51:47.159802Z","iopub.status.idle":"2024-01-10T10:51:48.184843Z","shell.execute_reply.started":"2024-01-10T10:51:47.159779Z","shell.execute_reply":"2024-01-10T10:51:48.183612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head /kaggle/working/small/train/labels/0582.txt","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:48.186436Z","iopub.execute_input":"2024-01-10T10:51:48.186765Z","iopub.status.idle":"2024-01-10T10:51:49.206244Z","shell.execute_reply.started":"2024-01-10T10:51:48.186738Z","shell.execute_reply":"2024-01-10T10:51:49.205180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train yolov8 model","metadata":{}},{"cell_type":"code","source":"project = 'hacking_human_vasculature_small'\nname = 'small_set'","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:51:49.208032Z","iopub.execute_input":"2024-01-10T10:51:49.208490Z","iopub.status.idle":"2024-01-10T10:51:49.214167Z","shell.execute_reply.started":"2024-01-10T10:51:49.208452Z","shell.execute_reply":"2024-01-10T10:51:49.213085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U ipywidgets","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:53:00.224937Z","iopub.execute_input":"2024-01-10T10:53:00.225330Z","iopub.status.idle":"2024-01-10T10:53:14.208317Z","shell.execute_reply.started":"2024-01-10T10:53:00.225299Z","shell.execute_reply":"2024-01-10T10:53:14.207038Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Enable wandb logging","metadata":{}},{"cell_type":"code","source":"import wandb\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nwandb_api_key = user_secrets.get_secret(\"wandb-api-key\")\nwandb.login(key=wandb_api_key)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:54:32.425413Z","iopub.execute_input":"2024-01-10T10:54:32.426353Z","iopub.status.idle":"2024-01-10T10:54:50.709902Z","shell.execute_reply.started":"2024-01-10T10:54:32.426309Z","shell.execute_reply":"2024-01-10T10:54:50.708923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load segmentation model weights from pretrained and train on the small set","metadata":{}},{"cell_type":"code","source":"from ultralytics import YOLO\n\nmodel = YOLO('yolov8n-seg.pt')","metadata":{"execution":{"iopub.status.busy":"2024-01-10T10:54:57.348486Z","iopub.execute_input":"2024-01-10T10:54:57.349254Z","iopub.status.idle":"2024-01-10T10:54:57.701601Z","shell.execute_reply.started":"2024-01-10T10:54:57.349219Z","shell.execute_reply":"2024-01-10T10:54:57.700569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.train(data='small/small_yolo.yaml',\n                      project=project,\n                      name=name,\n                      epochs=200,\n                      patience=0, #I am setting patience=0 to disable early stopping.\n                      batch=4,\n                      imgsz=1706,\n                      device=[0, 1]\n                     )","metadata":{"execution":{"iopub.status.busy":"2024-01-10T11:07:59.223020Z","iopub.execute_input":"2024-01-10T11:07:59.223464Z","iopub.status.idle":"2024-01-10T11:11:31.757641Z","shell.execute_reply.started":"2024-01-10T11:07:59.223432Z","shell.execute_reply":"2024-01-10T11:11:31.753514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2024-01-10T11:23:38.426869Z","iopub.execute_input":"2024-01-10T11:23:38.427311Z","iopub.status.idle":"2024-01-10T11:23:38.432705Z","shell.execute_reply.started":"2024-01-10T11:23:38.427278Z","shell.execute_reply":"2024-01-10T11:23:38.431653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}