{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Baseline Testing with YOLOv8\n\nThis is some of my initial work. Still a lot for me to learn in ML so any and all pointers would be appreciated ❤️","metadata":{}},{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image, ImageDraw","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-25T11:03:19.990018Z","iopub.execute_input":"2023-05-25T11:03:19.990422Z","iopub.status.idle":"2023-05-25T11:03:20.005530Z","shell.execute_reply.started":"2023-05-25T11:03:19.990395Z","shell.execute_reply":"2023-05-25T11:03:20.004558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Cfg: \n    BASE_PATH = Path('../input/hubmap-hacking-the-human-vasculature')\n    TRAIN_PATH = BASE_PATH / 'train'\n    TEST_PATH = BASE_PATH / 'test'\n    POLYGONS_PATH = BASE_PATH / 'polygons.jsonl'\n    \n    OUTPUT_ROOT = Path('/kaggle/working')\n    IMAGE_SIZE = 512\n    YOLO_TRAIN = OUTPUT_ROOT / 'yolo_data' / 'train'\n    YOLO_VAL = OUTPUT_ROOT / 'yolo_data'/ 'val'\n    \n    DATASET_CONFIG =  OUTPUT_ROOT / 'dataset.yaml'\n    MODEL_NAME = 'HUBMap-YOLOv8'\n    \n    IMAGE_SIZE=512\n    N_EPOCHS = 100\n    N_BATCH = 16\n    RANDOM_STATE = 2023\n    SAMPLE_SIZE = 1.0\n    TEST_SIZE = .2\n    INDEX = 'id'","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:55.745926Z","iopub.execute_input":"2023-05-25T11:04:55.746352Z","iopub.status.idle":"2023-05-25T11:04:55.755273Z","shell.execute_reply.started":"2023-05-25T11:04:55.746317Z","shell.execute_reply":"2023-05-25T11:04:55.754302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initial overview of the Data","metadata":{}},{"cell_type":"code","source":"df_meta = pd.read_csv(Cfg.BASE_PATH / 'tile_meta.csv')\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:22.660649Z","iopub.execute_input":"2023-05-25T11:03:22.661001Z","iopub.status.idle":"2023-05-25T11:03:22.727652Z","shell.execute_reply.started":"2023-05-25T11:03:22.660973Z","shell.execute_reply":"2023-05-25T11:03:22.726619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The competition data comprises tiles extracted from five Whole Slide Images (WSI) split into two datasets. Tiles from Dataset 1 have annotations that have been expert reviewed. Dataset 2 comprises the remaining tiles from these same WSIs and contain sparse annotations that have not been expert reviewed.\n\n> We also include, as Dataset 3, tiles extracted from an additional nine WSIs. These tiles have not been annotated. You may wish to apply semi- or self-supervised learning techniques on this data to support your predictions.\n\nWill probably be useful to figure out what split of data we have there then","metadata":{}},{"cell_type":"code","source":"df_meta_by_dataset = {i: grp for i, grp in df_meta.groupby('dataset')}\ndf_meta_by_dataset","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:25.614607Z","iopub.execute_input":"2023-05-25T11:03:25.615045Z","iopub.status.idle":"2023-05-25T11:03:25.662760Z","shell.execute_reply.started":"2023-05-25T11:03:25.615008Z","shell.execute_reply":"2023-05-25T11:03:25.661857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar([i[0] for i in df_meta_by_dataset.items()], [len(i[1]) for i in df_meta_by_dataset.items()])","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:27.590207Z","iopub.execute_input":"2023-05-25T11:03:27.590631Z","iopub.status.idle":"2023-05-25T11:03:27.957188Z","shell.execute_reply.started":"2023-05-25T11:03:27.590598Z","shell.execute_reply":"2023-05-25T11:03:27.956110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Looking at the JSONL annotations","metadata":{}},{"cell_type":"code","source":"df_polygons = pd.read_json(Cfg.POLYGONS_PATH, lines=True)\ndf_polygons","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:03:29.953132Z","iopub.execute_input":"2023-05-25T11:03:29.953568Z","iopub.status.idle":"2023-05-25T11:03:34.297533Z","shell.execute_reply.started":"2023-05-25T11:03:29.953530Z","shell.execute_reply":"2023-05-25T11:03:34.296512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setup YOLO Dataset Format","metadata":{}},{"cell_type":"code","source":"!mkdir yolo_data\n!mkdir yolo_data/val\n!mkdir yolo_data/train","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:38.092711Z","iopub.execute_input":"2023-05-25T11:03:38.093091Z","iopub.status.idle":"2023-05-25T11:03:41.146004Z","shell.execute_reply.started":"2023-05-25T11:03:38.093055Z","shell.execute_reply":"2023-05-25T11:03:41.144813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For a baseline I only want to use images from dataset 1 - expert annotated\ndf_ds1 = df_meta_by_dataset[1]\n# Make sure we only get the dataset 1 polygons too\npolydf_ds1 = df_polygons[df_polygons['id'].isin(df_ds1['id'])]\nprint(len(polydf_ds1))\nprint(len(df_ds1))","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:41.787308Z","iopub.execute_input":"2023-05-25T11:03:41.787699Z","iopub.status.idle":"2023-05-25T11:03:41.802517Z","shell.execute_reply.started":"2023-05-25T11:03:41.787668Z","shell.execute_reply":"2023-05-25T11:03:41.801317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polydf_ds1","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:43.581472Z","iopub.execute_input":"2023-05-25T11:03:43.581829Z","iopub.status.idle":"2023-05-25T11:03:43.678401Z","shell.execute_reply.started":"2023-05-25T11:03:43.581800Z","shell.execute_reply":"2023-05-25T11:03:43.677177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Need to Normalise the co-ords","metadata":{}},{"cell_type":"code","source":"def map_annotations(annot_class):\n    if(annot_class == \"blood_vessel\"):\n        return 0\n    elif (annot_class == \"glomerulus\"):\n        return 1\n    else:\n        #Unsure case\n        return 2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_annotation(annotation):\n#     return [{'type': map_annotations(a['type']), 'coordinates': [ np.array(coord)/512 for coord in a['coordinates'][0] ]} for a in annotation]\n# woops only one class\n    return [{'type': 0, 'coordinates': [ np.array(coord)/512 for coord in a['coordinates'][0] ]} for a in annotation]","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:48.512286Z","iopub.execute_input":"2023-05-25T11:03:48.512677Z","iopub.status.idle":"2023-05-25T11:03:48.518513Z","shell.execute_reply.started":"2023-05-25T11:03:48.512648Z","shell.execute_reply":"2023-05-25T11:03:48.517308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polydf_ds1.loc[:, 'norm_annotations'] = polydf_ds1['annotations'].apply(lambda x: normalize_annotation(x))\n\npolydf_ds1.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:03:54.807951Z","iopub.execute_input":"2023-05-25T11:03:54.808359Z","iopub.status.idle":"2023-05-25T11:04:00.900142Z","shell.execute_reply.started":"2023-05-25T11:03:54.808328Z","shell.execute_reply":"2023-05-25T11:04:00.898996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generate .txt files to accompany data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train, df_val = train_test_split(polydf_ds1, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:03.568419Z","iopub.execute_input":"2023-05-25T11:04:03.568775Z","iopub.status.idle":"2023-05-25T11:04:04.359002Z","shell.execute_reply.started":"2023-05-25T11:04:03.568745Z","shell.execute_reply":"2023-05-25T11:04:04.357981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polydf_ds1['norm_annotations'][2][0]['coordinates']","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:04:06.059486Z","iopub.execute_input":"2023-05-25T11:04:06.059889Z","iopub.status.idle":"2023-05-25T11:04:06.120542Z","shell.execute_reply.started":"2023-05-25T11:04:06.059858Z","shell.execute_reply":"2023-05-25T11:04:06.119580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Write the text files and copy across the relevant image files needed for training\ndef yolo_prep_text_data(path, id, annotations):\n    with open(f\"{path/id}.txt\", 'w+') as file:\n        #Need to write one line per annotation\n        for annot in annotations:\n            formatted_polys = \" \".join(str(x) for x in np.hstack(annot['coordinates']).tolist())\n            \n            file.write(f\"{annot['type']} {formatted_polys}\")\n            file.write(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:07.778593Z","iopub.execute_input":"2023-05-25T11:04:07.779225Z","iopub.status.idle":"2023-05-25T11:04:07.785707Z","shell.execute_reply.started":"2023-05-25T11:04:07.779188Z","shell.execute_reply":"2023-05-25T11:04:07.784483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n#Prep the validation files\nfor index, row in df_val.iterrows():\n    shutil.copyfile(f\"{Cfg.TRAIN_PATH}/{row['id']}.tif\",f\"{Cfg.YOLO_VAL}/{row['id']}.tif\")\n    yolo_prep_text_data(Cfg.YOLO_VAL, row['id'], row['norm_annotations'] )","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:04:09.599075Z","iopub.execute_input":"2023-05-25T11:04:09.600130Z","iopub.status.idle":"2023-05-25T11:04:11.490275Z","shell.execute_reply.started":"2023-05-25T11:04:09.600077Z","shell.execute_reply":"2023-05-25T11:04:11.489315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/yolo_data/val","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:04:12.430229Z","iopub.execute_input":"2023-05-25T11:04:12.430919Z","iopub.status.idle":"2023-05-25T11:04:13.551678Z","shell.execute_reply.started":"2023-05-25T11:04:12.430883Z","shell.execute_reply":"2023-05-25T11:04:13.550227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prep the train files\nfor index, row in df_train.iterrows():\n    shutil.copyfile(f\"{Cfg.TRAIN_PATH}/{row['id']}.tif\",f\"{Cfg.YOLO_TRAIN}/{row['id']}.tif\")\n    yolo_prep_text_data(Cfg.YOLO_TRAIN, row['id'], row['norm_annotations'] )","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:15.986060Z","iopub.execute_input":"2023-05-25T11:04:15.986906Z","iopub.status.idle":"2023-05-25T11:04:23.294388Z","shell.execute_reply.started":"2023-05-25T11:04:15.986852Z","shell.execute_reply":"2023-05-25T11:04:23.293368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/yolo_data/train","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:04:23.296165Z","iopub.execute_input":"2023-05-25T11:04:23.296507Z","iopub.status.idle":"2023-05-25T11:04:24.314047Z","shell.execute_reply.started":"2023-05-25T11:04:23.296480Z","shell.execute_reply":"2023-05-25T11:04:24.312848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Outline Dataset YAML","metadata":{}},{"cell_type":"code","source":"!cd /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:24.316261Z","iopub.execute_input":"2023-05-25T11:04:24.316723Z","iopub.status.idle":"2023-05-25T11:04:25.323659Z","shell.execute_reply.started":"2023-05-25T11:04:24.316683Z","shell.execute_reply":"2023-05-25T11:04:25.322348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile dataset.yaml\n\ntrain: yolo_data/train/\nval: yolo_data/val/\n\nnames:\n    0: vasculature","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:25.327003Z","iopub.execute_input":"2023-05-25T11:04:25.328064Z","iopub.status.idle":"2023-05-25T11:04:25.335449Z","shell.execute_reply.started":"2023-05-25T11:04:25.328006Z","shell.execute_reply":"2023-05-25T11:04:25.334427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Throw a YOLO at it and see what happens","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"execution":{"iopub.status.idle":"2023-05-25T11:04:47.791210Z","shell.execute_reply.started":"2023-05-25T11:04:28.245833Z","shell.execute_reply":"2023-05-25T11:04:47.789928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ultralytics\nfrom ultralytics import YOLO\nultralytics.checks()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:47.794422Z","iopub.execute_input":"2023-05-25T11:04:47.795381Z","iopub.status.idle":"2023-05-25T11:04:55.500810Z","shell.execute_reply.started":"2023-05-25T11:04:47.795327Z","shell.execute_reply":"2023-05-25T11:04:55.499684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO('yolov8m-seg')","metadata":{"execution":{"iopub.status.busy":"2023-05-25T11:04:59.172574Z","iopub.execute_input":"2023-05-25T11:04:59.172937Z","iopub.status.idle":"2023-05-25T11:04:59.335252Z","shell.execute_reply.started":"2023-05-25T11:04:59.172908Z","shell.execute_reply":"2023-05-25T11:04:59.334244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.train(\n   data=str(Cfg.DATASET_CONFIG),\n   epochs=Cfg.N_EPOCHS,\n   imgsz=Cfg.IMAGE_SIZE,\n   batch=Cfg.N_BATCH,\n   save=True, \n   verbose=False,\n   name=Cfg.MODEL_NAME)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T11:05:05.105375Z","iopub.execute_input":"2023-05-25T11:05:05.105760Z","iopub.status.idle":"2023-05-25T11:51:14.128225Z","shell.execute_reply.started":"2023-05-25T11:05:05.105729Z","shell.execute_reply":"2023-05-25T11:51:14.126853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!rm -r /kaggle/working/runs/segment/HUBMap-YOLOv82","metadata":{"execution":{"iopub.status.busy":"2023-05-24T20:18:00.944137Z","iopub.execute_input":"2023-05-24T20:18:00.944638Z","iopub.status.idle":"2023-05-24T20:18:02.225971Z","shell.execute_reply.started":"2023-05-24T20:18:00.944599Z","shell.execute_reply":"2023-05-24T20:18:02.223871Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_root = Cfg.OUTPUT_ROOT / 'runs/segment' / f\"{Cfg.MODEL_NAME}3\"\n!tree /kaggle/working/runs","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:20:58.951612Z","iopub.execute_input":"2023-05-25T12:20:58.952410Z","iopub.status.idle":"2023-05-25T12:21:00.148940Z","shell.execute_reply.started":"2023-05-25T12:20:58.952372Z","shell.execute_reply":"2023-05-25T12:21:00.147527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\nImage(filename=result_root / 'results.png', width=1000)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:02.458004Z","iopub.execute_input":"2023-05-25T12:21:02.458486Z","iopub.status.idle":"2023-05-25T12:21:02.493178Z","shell.execute_reply.started":"2023-05-25T12:21:02.458447Z","shell.execute_reply":"2023-05-25T12:21:02.491741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image(filename=result_root / 'labels.jpg', width=1000)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:05.293601Z","iopub.execute_input":"2023-05-25T12:21:05.293998Z","iopub.status.idle":"2023-05-25T12:21:05.318871Z","shell.execute_reply.started":"2023-05-25T12:21:05.293965Z","shell.execute_reply":"2023-05-25T12:21:05.317089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image(filename=result_root / 'confusion_matrix.png', width=800)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:12.302739Z","iopub.execute_input":"2023-05-25T12:21:12.303170Z","iopub.status.idle":"2023-05-25T12:21:12.316828Z","shell.execute_reply.started":"2023-05-25T12:21:12.303137Z","shell.execute_reply":"2023-05-25T12:21:12.315856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image(filename=result_root / 'train_batch0.jpg', height=800, width=1200)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:18.112203Z","iopub.execute_input":"2023-05-25T12:21:18.112950Z","iopub.status.idle":"2023-05-25T12:21:18.155683Z","shell.execute_reply.started":"2023-05-25T12:21:18.112915Z","shell.execute_reply":"2023-05-25T12:21:18.154469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"best_model = YOLO(result_root / 'weights/best.pt')","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:23.949083Z","iopub.execute_input":"2023-05-25T12:21:23.949511Z","iopub.status.idle":"2023-05-25T12:21:24.132697Z","shell.execute_reply.started":"2023-05-25T12:21:23.949479Z","shell.execute_reply":"2023-05-25T12:21:24.131357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_result = best_model(Cfg.BASE_PATH / 'test/72e40acccadf.tif', conf=0.4)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-05-25T12:21:25.049532Z","iopub.execute_input":"2023-05-25T12:21:25.050120Z","iopub.status.idle":"2023-05-25T12:21:25.813001Z","shell.execute_reply.started":"2023-05-25T12:21:25.050076Z","shell.execute_reply":"2023-05-25T12:21:25.811748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib.patches import Polygon","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:27.449229Z","iopub.execute_input":"2023-05-25T12:21:27.449963Z","iopub.status.idle":"2023-05-25T12:21:27.464134Z","shell.execute_reply.started":"2023-05-25T12:21:27.449927Z","shell.execute_reply":"2023-05-25T12:21:27.463134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\ntest_image = Image.open(Cfg.BASE_PATH / \"test/72e40acccadf.tif\")\n\nfig, ax = plt.subplots(1, 1, figsize=(12, 6))\n\n# Plot the image\nax.imshow(test_image)\nax.set_title(\"Image\")\n\nfor p in test_result[0].masks.xy:\n    polygon = Polygon(p, fill=True, facecolor=(0,0,1,0.2), edgecolor='b')\n    ax.add_patch(polygon)\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T12:21:59.167100Z","iopub.execute_input":"2023-05-25T12:21:59.167467Z","iopub.status.idle":"2023-05-25T12:21:59.774908Z","shell.execute_reply.started":"2023-05-25T12:21:59.167439Z","shell.execute_reply":"2023-05-25T12:21:59.773743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}