{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### YOLO v8 train & inference\n\nWe use the YOLO V8 model for this competition because it can execute the object detection and segmentation at the same time.  \nBecause of this notebook is online, we can't submit this directly.  ","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:11.343470Z","iopub.execute_input":"2023-05-29T05:14:11.343839Z","iopub.status.idle":"2023-05-29T05:14:32.167326Z","shell.execute_reply.started":"2023-05-29T05:14:11.343807Z","shell.execute_reply":"2023-05-29T05:14:32.166032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\nimport pandas as pd\nimport numpy as np\nimport tifffile as tiff\nimport matplotlib.pyplot as plt\n\nfrom pathlib import Path\nfrom glob import glob\nfrom collections import defaultdict\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nfrom IPython.display import Image as show_image\n\nimport ultralytics\nfrom ultralytics import YOLO\n\nimport torch\n\nultralytics.checks()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-29T05:41:18.389846Z","iopub.execute_input":"2023-05-29T05:41:18.390368Z","iopub.status.idle":"2023-05-29T05:41:18.415634Z","shell.execute_reply.started":"2023-05-29T05:41:18.390325Z","shell.execute_reply":"2023-05-29T05:41:18.414445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set parameters","metadata":{}},{"cell_type":"markdown","source":"### Hyper parameters","metadata":{}},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\nIMAGE_SIZE = 512\nBATCH_SIZE = 16\nEPOCHS = 10\n\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:39.011051Z","iopub.execute_input":"2023-05-29T05:14:39.011600Z","iopub.status.idle":"2023-05-29T05:14:39.021196Z","shell.execute_reply.started":"2023-05-29T05:14:39.011570Z","shell.execute_reply":"2023-05-29T05:14:39.020040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Directories","metadata":{}},{"cell_type":"code","source":"def mkdir_yolo_data(train_path, val_path):\n    \"\"\"\n    make yolo data's directories\n    \n    parameters\n    ----------\n    train_path: str\n        path for training data\n    val_path: str\n        path for validation data\n    \n    returns\n    ----------\n    train_image_path: str\n        path for images of training data\n    train_label_path: str\n        path for labels of trainingdata\n    val_image_path: str\n        path for images of validation data\n    val_label_path: str\n        path for labels of validation data\n    \"\"\"\n    train_image_path = Path(f'{train_path}/images')\n    train_label_path = Path(f'{train_path}/labels')\n    val_image_path = Path(f'{val_path}/images')\n    val_label_path = Path(f'{val_path}/labels')\n    \n    train_image_path.mkdir(parents=True, exist_ok=True)\n    train_label_path.mkdir(parents=True, exist_ok=True)\n    val_image_path.mkdir(parents=True, exist_ok=True)\n    val_label_path.mkdir(parents=True, exist_ok=True)\n    \n    return train_image_path, train_label_path, val_image_path, val_label_path","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-29T05:14:39.025445Z","iopub.execute_input":"2023-05-29T05:14:39.026258Z","iopub.status.idle":"2023-05-29T05:14:39.035160Z","shell.execute_reply.started":"2023-05-29T05:14:39.026229Z","shell.execute_reply":"2023-05-29T05:14:39.033936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# File path settings\nBASE_DIR = Path('/kaggle/input/hubmap-hacking-the-human-vasculature')\n\ntest_paths = glob(f'{BASE_DIR}/test/*')\npolygons_path = f'{BASE_DIR}/polygons.jsonl'\n\nyolo_train_path = 'datasets/train'\nyolo_val_path = 'datasets/val'","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:39.036457Z","iopub.execute_input":"2023-05-29T05:14:39.036807Z","iopub.status.idle":"2023-05-29T05:14:39.054896Z","shell.execute_reply.started":"2023-05-29T05:14:39.036780Z","shell.execute_reply":"2023-05-29T05:14:39.053520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make directories\ntrain_image_path, train_label_path, \\\n    val_image_path, val_label_path = mkdir_yolo_data(yolo_train_path, yolo_val_path)\nprint(train_image_path)\nprint(train_label_path)\nprint(val_image_path)\nprint(val_label_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:39.056663Z","iopub.execute_input":"2023-05-29T05:14:39.057782Z","iopub.status.idle":"2023-05-29T05:14:39.065358Z","shell.execute_reply.started":"2023-05-29T05:14:39.057742Z","shell.execute_reply":"2023-05-29T05:14:39.064036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create annotation files and move tif to yolo' directory","metadata":{}},{"cell_type":"code","source":"def create_vessel_annotations(polygons_path):\n    \"\"\"\n    Create annotations set which have blood_vessel label.\n    \n    parameters\n    ----------\n    polygons_path: str\n        path of polygons.jsonl\n    \n    returns\n    ----------\n    annotations_dict: dict {key=id, value=coordinates}\n        annotations dict with key id and value coordinates of blood_vessel\n    \"\"\"\n    # load polygons data\n    polygons = pd.read_json(polygons_path, orient='records', lines=True)\n    \n    # extract blood_vessel annotation\n    annotations_dict = defaultdict(list)\n    for idx, row in polygons.iterrows():\n        id_ = row['id']\n        annotations = row['annotations']\n        for annotation in annotations:\n            if annotation['type'] == 'blood_vessel':\n                annotations_dict[id_].append(annotation['coordinates'])\n    \n    return annotations_dict\n\ndef create_label_file(id_, coordinates, path):\n    \"\"\"\n    Create label txt file for yolo v8\n    \n    parameters\n    ----------\n    id_: str\n        label id\n    coordinates: list\n        coordinates of blood_vessel\n    path: str\n        path for saving label txt file\n    \"\"\"\n    label_txt = ''\n    for coordinate in coordinates:\n        label_txt += '0 '\n        # Normalize\n        coor_array = np.array(coordinate[0]).astype(float)\n        coor_array /= float(IMAGE_SIZE)\n        # transform to str\n        coor_list = list(coor_array.reshape(-1).astype(str))\n        coor_str = ' '.join(coor_list)\n        # add string to label txt\n        label_txt += f'{coor_str}\\n'\n    \n    # Write labels to txt file\n    with open(f'{path}/{id_}.txt', 'w') as f:\n        f.write(label_txt)\n        \ndef prepare_yolo_dataset(\n        annotaions_dict, train_image_path, train_label_path, \n        val_image_path, val_label_path):\n    \"\"\"\n    Prepare yolo dataset with images and labels\n    \n    parameters\n    ----------\n    annotations_dict: dict {key=id, value=coordinates}\n        annotations dict with key id and value coordinates of blood_vessel\n    train_image_path: str\n        path for images of training data\n    train_label_path: str\n        path for labels of trainingdata\n    val_image_path: str\n        path for images of validation data\n    val_label_path: str\n        path for labels of validation data\n    \"\"\"\n    ids = list(annotations_dict.keys())\n    \n    # train test split\n    indices = [i for i in range(len(ids))]\n    train_indices, val_indices = train_test_split(indices, test_size=0.2, random_state=1234)\n    \n    # Training data\n    for index in tqdm(train_indices):\n        id_ = ids[index]\n        \n        # create label txt file\n        create_label_file(id_, annotations_dict[id_], train_label_path)\n        # copy tif image file to yolo directory\n        source_file = f'{BASE_DIR}/train/{id_}.tif'\n        shutil.copy2(source_file, train_image_path)\n    \n    # Validation data\n    for index in tqdm(val_indices):\n        id_ = ids[index]\n        \n        # create label txt file\n        create_label_file(id_, annotations_dict[id_], val_label_path)\n        # copy tif image file to yolo directory\n        source_file = f'{BASE_DIR}/train/{id_}.tif'\n        shutil.copy2(source_file, val_image_path)\n    ","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-05-29T05:14:39.067468Z","iopub.execute_input":"2023-05-29T05:14:39.068369Z","iopub.status.idle":"2023-05-29T05:14:39.086640Z","shell.execute_reply.started":"2023-05-29T05:14:39.068331Z","shell.execute_reply":"2023-05-29T05:14:39.085393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create annotations dict with key=id and value=coordinates\nannotations_dict = create_vessel_annotations(polygons_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:39.088572Z","iopub.execute_input":"2023-05-29T05:14:39.089394Z","iopub.status.idle":"2023-05-29T05:14:44.171277Z","shell.execute_reply.started":"2023-05-29T05:14:39.089352Z","shell.execute_reply":"2023-05-29T05:14:44.170194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare dataset for yolo training\nprepare_yolo_dataset(\n    annotations_dict, train_image_path, train_label_path,\n    val_image_path, val_label_path\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:14:44.173001Z","iopub.execute_input":"2023-05-29T05:14:44.173409Z","iopub.status.idle":"2023-05-29T05:15:19.950275Z","shell.execute_reply.started":"2023-05-29T05:14:44.173372Z","shell.execute_reply":"2023-05-29T05:15:19.948610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## YOLO","metadata":{}},{"cell_type":"code","source":"# Edit yaml content\nyaml_content = f'''\ntrain: train/images\nval: val/images\n\nnames:\n    0: blood_vessel\n'''\n\nyaml_file = 'data.yaml'\n\nwith open(yaml_file, 'w') as f:\n    f.write(yaml_content)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:15:19.956585Z","iopub.execute_input":"2023-05-29T05:15:19.956944Z","iopub.status.idle":"2023-05-29T05:15:19.963119Z","shell.execute_reply.started":"2023-05-29T05:15:19.956915Z","shell.execute_reply":"2023-05-29T05:15:19.961701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare model\nmodel = YOLO('yolov8n-seg.pt')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:15:19.964895Z","iopub.execute_input":"2023-05-29T05:15:19.965612Z","iopub.status.idle":"2023-05-29T05:15:21.097296Z","shell.execute_reply.started":"2023-05-29T05:15:19.965575Z","shell.execute_reply":"2023-05-29T05:15:21.096057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training\nresults = model.train(\n    batch=BATCH_SIZE,\n    device=0,\n    data=yaml_file,\n    epochs=EPOCHS,\n    imgsz=IMAGE_SIZE\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:18:52.510373Z","iopub.execute_input":"2023-05-29T05:18:52.510789Z","iopub.status.idle":"2023-05-29T05:37:33.580437Z","shell.execute_reply.started":"2023-05-29T05:18:52.510757Z","shell.execute_reply":"2023-05-29T05:37:33.579008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls runs/segment/train","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:37:43.439334Z","iopub.execute_input":"2023-05-29T05:37:43.439750Z","iopub.status.idle":"2023-05-29T05:37:44.634350Z","shell.execute_reply.started":"2023-05-29T05:37:43.439711Z","shell.execute_reply":"2023-05-29T05:37:44.632644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(filename='runs/segment/train/val_batch0_pred.jpg')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:37:44.637366Z","iopub.execute_input":"2023-05-29T05:37:44.641755Z","iopub.status.idle":"2023-05-29T05:37:44.693650Z","shell.execute_reply.started":"2023-05-29T05:37:44.641701Z","shell.execute_reply":"2023-05-29T05:37:44.687099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(filename='runs/segment/train/val_batch0_labels.jpg')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:37:48.384026Z","iopub.execute_input":"2023-05-29T05:37:48.385254Z","iopub.status.idle":"2023-05-29T05:37:48.418312Z","shell.execute_reply.started":"2023-05-29T05:37:48.385201Z","shell.execute_reply":"2023-05-29T05:37:48.417135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(filename='runs/segment/train/results.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:37:48.866935Z","iopub.execute_input":"2023-05-29T05:37:48.868064Z","iopub.status.idle":"2023-05-29T05:37:48.897399Z","shell.execute_reply.started":"2023-05-29T05:37:48.868017Z","shell.execute_reply":"2023-05-29T05:37:48.896102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(filename='runs/segment/train/MaskP_curve.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:37:49.613258Z","iopub.execute_input":"2023-05-29T05:37:49.614042Z","iopub.status.idle":"2023-05-29T05:37:49.629083Z","shell.execute_reply.started":"2023-05-29T05:37:49.613997Z","shell.execute_reply":"2023-05-29T05:37:49.627597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trained_model = YOLO('runs/segment/train/weights/best.pt')\nresults = list(trained_model.predict(test_paths, save=True, conf=0.6))\nresult = results[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:43:40.227739Z","iopub.execute_input":"2023-05-29T05:43:40.228156Z","iopub.status.idle":"2023-05-29T05:43:40.495495Z","shell.execute_reply.started":"2023-05-29T05:43:40.228123Z","shell.execute_reply":"2023-05-29T05:43:40.494221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls runs/segment/predict","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:42:49.348244Z","iopub.execute_input":"2023-05-29T05:42:49.348651Z","iopub.status.idle":"2023-05-29T05:42:50.593025Z","shell.execute_reply.started":"2023-05-29T05:42:49.348618Z","shell.execute_reply":"2023-05-29T05:42:50.591480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = tiff.imread('runs/segment/predict/72e40acccadf.tif')\nplt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:43:50.208951Z","iopub.execute_input":"2023-05-29T05:43:50.209381Z","iopub.status.idle":"2023-05-29T05:43:50.752678Z","shell.execute_reply.started":"2023-05-29T05:43:50.209346Z","shell.execute_reply":"2023-05-29T05:43:50.749664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}