{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":15.865553,"end_time":"2023-10-17T02:44:10.768626","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2023-10-17T02:43:54.903073","version":"2.4.0"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52279,"databundleVersionId":5822112,"sourceType":"competition"}],"dockerImageVersionId":30616,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics jsonline","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:57:44.491019Z","iopub.execute_input":"2023-12-10T13:57:44.491364Z","iopub.status.idle":"2023-12-10T13:57:58.002667Z","shell.execute_reply.started":"2023-12-10T13:57:44.491333Z","shell.execute_reply":"2023-12-10T13:57:58.001727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\nimport json\nfrom colorama import Fore","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":2.236316,"end_time":"2023-10-17T02:44:00.615569","exception":false,"start_time":"2023-10-17T02:43:58.379253","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.004412Z","iopub.execute_input":"2023-12-10T13:57:58.004752Z","iopub.status.idle":"2023-12-10T13:57:58.838286Z","shell.execute_reply.started":"2023-12-10T13:57:58.004721Z","shell.execute_reply":"2023-12-10T13:57:58.837368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    image_train_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\n    image_test_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\n    wsi_meta_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\"\n    tile_meta_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\"\n    polygons_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\n    output_path = \"/kaggle/working/\"","metadata":{"papermill":{"duration":0.012591,"end_time":"2023-10-17T02:44:00.634310","exception":false,"start_time":"2023-10-17T02:44:00.621719","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.839575Z","iopub.execute_input":"2023-12-10T13:57:58.840053Z","iopub.status.idle":"2023-12-10T13:57:58.845463Z","shell.execute_reply.started":"2023-12-10T13:57:58.840017Z","shell.execute_reply":"2023-12-10T13:57:58.844454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thông tin về dữ liệu:\n\nwsi_meta: metadata cho mỗi loại wsi source (Whole Slide Images) đặc trưng của mỗi loại wsi\n\ntile_meta: metadata cho mỗi ảnh","metadata":{"papermill":{"duration":0.003866,"end_time":"2023-10-17T02:44:00.642334","exception":false,"start_time":"2023-10-17T02:44:00.638468","status":"completed"},"tags":[]}},{"cell_type":"code","source":"wsi_df = pd.read_csv(config.wsi_meta_path)\nwsi_df","metadata":{"papermill":{"duration":0.04483,"end_time":"2023-10-17T02:44:00.691616","exception":false,"start_time":"2023-10-17T02:44:00.646786","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.847814Z","iopub.execute_input":"2023-12-10T13:57:58.848077Z","iopub.status.idle":"2023-12-10T13:57:58.941112Z","shell.execute_reply.started":"2023-12-10T13:57:58.848053Z","shell.execute_reply":"2023-12-10T13:57:58.940230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df.info()","metadata":{"papermill":{"duration":0.03934,"end_time":"2023-10-17T02:44:00.735414","exception":false,"start_time":"2023-10-17T02:44:00.696074","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.942282Z","iopub.execute_input":"2023-12-10T13:57:58.943091Z","iopub.status.idle":"2023-12-10T13:57:58.965078Z","shell.execute_reply.started":"2023-12-10T13:57:58.943054Z","shell.execute_reply":"2023-12-10T13:57:58.964191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df.describe()","metadata":{"papermill":{"duration":0.033314,"end_time":"2023-10-17T02:44:00.773258","exception":false,"start_time":"2023-10-17T02:44:00.739944","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.966020Z","iopub.execute_input":"2023-12-10T13:57:58.966258Z","iopub.status.idle":"2023-12-10T13:57:58.990859Z","shell.execute_reply.started":"2023-12-10T13:57:58.966235Z","shell.execute_reply":"2023-12-10T13:57:58.990025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in wsi_df.select_dtypes(include=['int64', 'float64']).columns:\n    plt.figure()\n    sns.distplot(wsi_df[col])\n    plt.title(f'Distribution of {col}')","metadata":{"papermill":{"duration":1.746005,"end_time":"2023-10-17T02:44:02.524107","exception":false,"start_time":"2023-10-17T02:44:00.778102","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:57:58.991882Z","iopub.execute_input":"2023-12-10T13:57:58.992200Z","iopub.status.idle":"2023-12-10T13:58:00.517777Z","shell.execute_reply.started":"2023-12-10T13:57:58.992165Z","shell.execute_reply":"2023-12-10T13:58:00.516895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in wsi_df.select_dtypes(include='object').columns:\n    plt.figure() \n    sns.countplot(x=col, data=wsi_df)\n    plt.title(f'Distribution of {col}')","metadata":{"papermill":{"duration":0.494371,"end_time":"2023-10-17T02:44:03.027158","exception":false,"start_time":"2023-10-17T02:44:02.532787","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:00.519026Z","iopub.execute_input":"2023-12-10T13:58:00.519357Z","iopub.status.idle":"2023-12-10T13:58:00.973709Z","shell.execute_reply.started":"2023-12-10T13:58:00.519324Z","shell.execute_reply":"2023-12-10T13:58:00.972827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df = pd.read_csv(config.tile_meta_path)\ntile_df","metadata":{"papermill":{"duration":0.058891,"end_time":"2023-10-17T02:44:03.094478","exception":false,"start_time":"2023-10-17T02:44:03.035587","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:00.974971Z","iopub.execute_input":"2023-12-10T13:58:00.975294Z","iopub.status.idle":"2023-12-10T13:58:01.003396Z","shell.execute_reply.started":"2023-12-10T13:58:00.975268Z","shell.execute_reply":"2023-12-10T13:58:01.002521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Số lượng dataset","metadata":{"papermill":{"duration":0.008533,"end_time":"2023-10-17T02:44:03.111750","exception":false,"start_time":"2023-10-17T02:44:03.103217","status":"completed"},"tags":[]}},{"cell_type":"code","source":"np.unique(tile_df.dataset)","metadata":{"papermill":{"duration":0.019846,"end_time":"2023-10-17T02:44:03.140068","exception":false,"start_time":"2023-10-17T02:44:03.120222","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:01.006672Z","iopub.execute_input":"2023-12-10T13:58:01.006940Z","iopub.status.idle":"2023-12-10T13:58:01.013284Z","shell.execute_reply.started":"2023-12-10T13:58:01.006916Z","shell.execute_reply":"2023-12-10T13:58:01.012203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"số lượng của loại wsi trong mỗi dataset","metadata":{"papermill":{"duration":0.008833,"end_time":"2023-10-17T02:44:03.157909","exception":false,"start_time":"2023-10-17T02:44:03.149076","status":"completed"},"tags":[]}},{"cell_type":"code","source":"\ntile_df[tile_df['dataset']==1]['source_wsi'].value_counts()","metadata":{"papermill":{"duration":0.01982,"end_time":"2023-10-17T02:44:03.186174","exception":false,"start_time":"2023-10-17T02:44:03.166354","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:01.014438Z","iopub.execute_input":"2023-12-10T13:58:01.014791Z","iopub.status.idle":"2023-12-10T13:58:01.028488Z","shell.execute_reply.started":"2023-12-10T13:58:01.014763Z","shell.execute_reply":"2023-12-10T13:58:01.027557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df[tile_df['dataset']==2]['source_wsi'].value_counts()","metadata":{"papermill":{"duration":0.021285,"end_time":"2023-10-17T02:44:03.216127","exception":false,"start_time":"2023-10-17T02:44:03.194842","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:01.029621Z","iopub.execute_input":"2023-12-10T13:58:01.030070Z","iopub.status.idle":"2023-12-10T13:58:01.042448Z","shell.execute_reply.started":"2023-12-10T13:58:01.030031Z","shell.execute_reply":"2023-12-10T13:58:01.041727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Các wsi trong dataset 3 chưa được miêu tả các đặc trưng -> không biết đặc trưng là gì -> không nên dùng để train","metadata":{"papermill":{"duration":0.008209,"end_time":"2023-10-17T02:44:03.233003","exception":false,"start_time":"2023-10-17T02:44:03.224794","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tile_df[tile_df['dataset']==3]['source_wsi'].value_counts()","metadata":{"papermill":{"duration":0.020018,"end_time":"2023-10-17T02:44:03.261499","exception":false,"start_time":"2023-10-17T02:44:03.241481","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:01.043536Z","iopub.execute_input":"2023-12-10T13:58:01.043789Z","iopub.status.idle":"2023-12-10T13:58:01.056142Z","shell.execute_reply.started":"2023-12-10T13:58:01.043765Z","shell.execute_reply":"2023-12-10T13:58:01.055373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ta chỉ có đặc trưng của 4 loại wsi là 1,2,3,4 -> quan sát tần suất của 4 loại này","metadata":{"papermill":{"duration":0.008456,"end_time":"2023-10-17T02:44:03.278684","exception":false,"start_time":"2023-10-17T02:44:03.270228","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import plotly.express as px\nx = tile_df[tile_df['source_wsi'].isin([1,2,3,4])]['source_wsi'].value_counts()\nfig = px.bar(x = x.index,y = x.values)\nfig.update_layout(\n    title ={\n        'text': 'Tần suất của mỗi loại WSI source',\n        'y':0.95,\n        'x':0.5,\n        \n    },\n    xaxis_title=\"Frequency\", yaxis_title=\"Source WSI\"\n)\nfig.show()","metadata":{"papermill":{"duration":2.986282,"end_time":"2023-10-17T02:44:06.273532","exception":false,"start_time":"2023-10-17T02:44:03.287250","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:01.057153Z","iopub.execute_input":"2023-12-10T13:58:01.057447Z","iopub.status.idle":"2023-12-10T13:58:03.319128Z","shell.execute_reply.started":"2023-12-10T13:58:01.057420Z","shell.execute_reply":"2023-12-10T13:58:03.318208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tóm gọn - Dataset Overview:\n\n    - Cuộc thi này tập trung vào việc phân tích các slide mô học của mô thận người. Mục tiêu là xác định và định vị các cấu trúc vi mạch, đặc biệt là các mạch máu (blood li), trong các slide này.\n\n    - Bộ dữ liệu được chia thành ba bộ chính: Bộ dữ liệu 1, Bộ dữ liệu 2 và Bộ dữ liệu 3.\n\n    - metadata được cung cấp là brick_meta.csv và wsi_meta.csv. Những tệp này chứa thông tin về hình ảnh, bao gồm cả chi tiết nhân khẩu học về người hiến mô.\n\n    - Dữ liệu được cung cấp dưới dạng hình ảnh TIFF, với mỗi ô có kích thước 512x512.\n\n    - Vị trí segment chỉ có cho Tập dữ liệu 1 và Tập dữ liệu 2. Những mặt nạ này cung cấp chú thích (blood vessels, glomerulus) chi tiết cho từng hình ảnh.\n\n\n    - Nhãn cần dự đoán là các khu vực chứa mạch máu. Cuộc thi yêu cầu phát triển một thuật toán để tự động xác định và định vị các cấu trúc mạch máu trên các lát mảnh thận người.","metadata":{}},{"cell_type":"code","source":"pip install jsonlines","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:03.320103Z","iopub.execute_input":"2023-12-10T13:58:03.320351Z","iopub.status.idle":"2023-12-10T13:58:15.357082Z","shell.execute_reply.started":"2023-12-10T13:58:03.320327Z","shell.execute_reply":"2023-12-10T13:58:15.355963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jsonlines\nwith open(config.polygons_path,'r') as f:\n    polygons_df= [line for line in jsonlines.Reader(f)]\n    \n","metadata":{"papermill":{"duration":2.420866,"end_time":"2023-10-17T02:44:08.703736","exception":false,"start_time":"2023-10-17T02:44:06.282870","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:15.358860Z","iopub.execute_input":"2023-12-10T13:58:15.359259Z","iopub.status.idle":"2023-12-10T13:58:18.410864Z","shell.execute_reply.started":"2023-12-10T13:58:15.359221Z","shell.execute_reply":"2023-12-10T13:58:18.410049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(polygons_df[:1])","metadata":{"papermill":{"duration":0.400616,"end_time":"2023-10-17T02:44:09.535741","exception":false,"start_time":"2023-10-17T02:44:09.135125","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-10T13:58:18.412142Z","iopub.execute_input":"2023-12-10T13:58:18.413018Z","iopub.status.idle":"2023-12-10T13:58:18.419981Z","shell.execute_reply.started":"2023-12-10T13:58:18.412978Z","shell.execute_reply":"2023-12-10T13:58:18.419126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.DataFrame(polygons_df)\n\nfor annotations in df['annotations'][0]:\n    coordinates = annotations['coordinates'][0]\n    label = annotations['type']\n    #flatten coordinates\n    print(f\"label: {label}, coordinates: {coordinates}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:18.421229Z","iopub.execute_input":"2023-12-10T13:58:18.421788Z","iopub.status.idle":"2023-12-10T13:58:18.441196Z","shell.execute_reply.started":"2023-12-10T13:58:18.421753Z","shell.execute_reply":"2023-12-10T13:58:18.440374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading image","metadata":{}},{"cell_type":"code","source":"import glob\nimport os\n\nall_img_path = glob.glob(os.path.join(config.image_train_path, \"*\"))\nfig,ax = plt.subplots(2,3,figsize=(15,15))\nfor i in range(2):\n    for j in range(3):\n        img_path = all_img_path[i*3 + j]\n        img = plt.imread(img_path)\n        ax[i][j].imshow(img)\n        ax[i][j].set_title(os.path.basename(img_path))\n        ax[i][j].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:18.442344Z","iopub.execute_input":"2023-12-10T13:58:18.442685Z","iopub.status.idle":"2023-12-10T13:58:20.111541Z","shell.execute_reply.started":"2023-12-10T13:58:18.442641Z","shell.execute_reply":"2023-12-10T13:58:20.110385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading label on image to see the blood vessel, glomerulus","metadata":{}},{"cell_type":"code","source":"#convert list of dict to dataframe\nimport pandas as pd\n\ndf = pd.DataFrame(polygons_df)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:20.112983Z","iopub.execute_input":"2023-12-10T13:58:20.113307Z","iopub.status.idle":"2023-12-10T13:58:20.218269Z","shell.execute_reply.started":"2023-12-10T13:58:20.113277Z","shell.execute_reply":"2023-12-10T13:58:20.217392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[:3]","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:20.219410Z","iopub.execute_input":"2023-12-10T13:58:20.219717Z","iopub.status.idle":"2023-12-10T13:58:20.244412Z","shell.execute_reply.started":"2023-12-10T13:58:20.219688Z","shell.execute_reply":"2023-12-10T13:58:20.243571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Thống kê các loại annotation trong tập dữ liệu\nannotation_counts = {}\nfor annotations in df[\"annotations\"]:\n    for annotation in annotations:\n        annotation_type = annotation['type']\n        if annotation_type in annotation_counts:\n            annotation_counts[annotation_type] += 1\n        else:\n            annotation_counts[annotation_type] = 1\n\nprint(annotation_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:20.245534Z","iopub.execute_input":"2023-12-10T13:58:20.245934Z","iopub.status.idle":"2023-12-10T13:58:20.266730Z","shell.execute_reply.started":"2023-12-10T13:58:20.245897Z","shell.execute_reply":"2023-12-10T13:58:20.265912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(list(annotation_counts.keys()), list(annotation_counts.values()), color=['r', 'g', 'b'])","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:20.267810Z","iopub.execute_input":"2023-12-10T13:58:20.268075Z","iopub.status.idle":"2023-12-10T13:58:20.428950Z","shell.execute_reply.started":"2023-12-10T13:58:20.268050Z","shell.execute_reply":"2023-12-10T13:58:20.428079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"==> Blood vessel chiếm đa số trong các ảnh","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(2, 3, figsize=(12, 8))\nsample = random.sample(range(len(df)), 6)\n\nfor i, ax in enumerate(axs.flatten()):\n    idx = sample[i]\n    annotations = df[\"annotations\"][idx]\n    img_path = os.path.join(config.image_train_path, f\"{df['id'][idx]}.tif\")\n    img = plt.imread(img_path)\n    ax.imshow(img)\n    ax.axis('off')\n\n    for annotation in annotations:\n        coordinates = annotation['coordinates']\n        annotation_type = annotation['type']\n\n        # Define colors based on annotation types\n        if annotation_type == 'glomerulus':\n            color = 'r'\n        elif annotation_type == 'blood_vessel':\n            color = 'y'\n        else:\n            color = 'b'\n\n        # Draw annotation on image\n        polygon = plt.Polygon(coordinates[0], linewidth=1, edgecolor=color, facecolor='none')\n        ax.add_patch(polygon)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:20.430340Z","iopub.execute_input":"2023-12-10T13:58:20.430904Z","iopub.status.idle":"2023-12-10T13:58:21.777231Z","shell.execute_reply.started":"2023-12-10T13:58:20.430867Z","shell.execute_reply":"2023-12-10T13:58:21.775818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tuy nhiên xét về mặt diện tích, đa phần các ảnh có diện tích của glomerulus lớn hơn diện tích của blood vessel, nên ta có thể xem xét việc dự đoán glomerulus có thể dễ dàng hơn dự đoán blood vessel.","metadata":{}},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:21.778289Z","iopub.status.idle":"2023-12-10T13:58:21.778632Z","shell.execute_reply.started":"2023-12-10T13:58:21.778465Z","shell.execute_reply":"2023-12-10T13:58:21.778482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"area_annotations = {\n    'glomerulus': [],\n    'blood_vessel': [],\n    'unsure': [],\n    \n}\n\nfor annotations in df[\"annotations\"]:\n    for annotation in annotations:\n        annotation_type = annotation['type']\n        coordinates = annotation['coordinates']\n        area = 0\n        for coordinate in coordinates:\n            area += cv2.contourArea(np.array(coordinate))\n        \n        if annotation_type in area_annotations:\n            area_annotations[annotation_type].append(area)\n\n# Chuyển đổi diện tích về đơn vị cm^2\narea_annotations = {k: sum(v) / len(v) for k, v in area_annotations.items()}\n\nprint(f\"Kích thước trung bình của các annotations:\")\nfor annotation_type, area in area_annotations.items():\n    print(f\"{annotation_type}: {area} (cm^2)\")","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:21.780086Z","iopub.status.idle":"2023-12-10T13:58:21.780422Z","shell.execute_reply.started":"2023-12-10T13:58:21.780258Z","shell.execute_reply":"2023-12-10T13:58:21.780274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\n# Lấy các loại annotations và diện tích tương ứng\nannotation_types = list(area_annotations.keys())\nareas = list(area_annotations.values())\n\n# Vẽ biểu đồ\nfig, ax = plt.subplots(1,2,figsize=(15,5))\nax[0].scatter(annotation_types, areas, s=areas, alpha=0.5, color=['r', 'g', 'b'])\n\n# Đặt các nhãn\nax[0].set_xlabel('Loại Annotation')\nax[0].set_ylabel('Diện tích (cm^2)')\nax[0].set_title('Biểu đồ diện tích của các annotations')\n\n#count số lượng\nax[1].bar(list(annotation_counts.keys()), list(annotation_counts.values()), color=['r', 'g', 'b'])\nax[1].set_xlabel('Loại Annotation')\nax[1].set_ylabel('Số lượng')\n\nprint(f\"Số lượng annotations mỗi loại: {annotation_counts}\")\n\n\n# Hiển thị biểu đồ\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:21.781975Z","iopub.status.idle":"2023-12-10T13:58:21.782420Z","shell.execute_reply.started":"2023-12-10T13:58:21.782191Z","shell.execute_reply":"2023-12-10T13:58:21.782213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make datasets","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nfrom colorama import Fore\nfrom tqdm import tqdm\nimport shutil\nimport yaml\nfrom itertools import chain\nimport numpy as np\n\nclass Dataset:\n    def __init__(self, images_dirpath: str, annotations_filepath: str, length: int = 1633):\n        self.train_size = None\n        self.val_size = None\n        self.length = length\n        self.classes = None\n        self.labels_counter = None\n        self.normalize = None\n        \n        self.images_dirpath = images_dirpath\n        self.annotations_filepath = annotations_filepath\n        self.dataset_dirpath = os.path.join(os.getcwd(), \"dataset\")\n        self.train_dirpath =  os.path.join(self.dataset_dirpath, \"train\")\n        self.val_dirpath =  os.path.join(self.dataset_dirpath, \"val\")\n        self.config_path = os.path.join(self.dataset_dirpath, \"config.yaml\")\n\n        self.samples = self.parse_jsonl(annotations_filepath)\n        self.classes_dict = {\n            \"blood_vessel\": 0,\n            \"glomerulus\": 1,\n            \"unsure\": 2,\n        }\n\n    def __prepare_dirs(self) -> None:\n        if os.path.exists(self.dataset_dirpath):\n            os.removedirs(self.dataset_dirpath)\n        os.makedirs(os.path.join(self.train_dirpath, \"images\"), exist_ok=True)\n        os.makedirs(os.path.join(self.train_dirpath, \"labels\"), exist_ok=True)\n        os.makedirs(os.path.join(self.val_dirpath, \"images\"), exist_ok=True)\n        os.makedirs(os.path.join(self.val_dirpath, \"labels\"), exist_ok=True)\n            # raise RuntimeError(\"Dataset already exists!\")\n\n    def __define_splitratio(self) -> None:\n        self.train_size = round(self.length * self.train_size)\n        self.val_size = self.length - self.train_size\n        assert self.train_size + self.val_size == self.length\n\n    def parse_jsonl(self, path: str):\n        with open(path, 'r') as json_file:\n            jsonl_samples = [\n                json.loads(line)\n                for line in tqdm(\n                    json_file, desc=\"Processing polygons\", total=self.length\n                )\n            ]\n        return jsonl_samples\n\n    def __define_paths(self, i: int) -> dict:\n        data_path = self.val_dirpath\n        if i < self.train_size:\n            data_path = self.train_dirpath\n        return {\n            \"images\": os.path.join(data_path, \"images\"),\n            \"labels\": os.path.join(data_path, \"labels\")\n        }\n\n    @staticmethod\n    def __get_label_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"labels\"],\n            f\"{identifier}.txt\"\n        )\n\n    @staticmethod\n    def __get_image_path(paths_dict: dict, identifier: str) -> str:\n        return os.path.join(\n            paths_dict[\"images\"],\n            f\"{identifier}.tif\"\n        )\n\n    def __copy_image(self, dst_path: str, identifier: str) -> str:\n        shutil.copyfile(\n            os.path.join(self.images_dirpath, f\"{identifier}.tif\"),\n            dst_path\n        )\n\n    def __copy_label(self, annotations: list, dst_path: str) -> None:\n        with open(dst_path, \"w\") as file:\n            for annotation in annotations:\n                coordinates = annotation[\"coordinates\"][0]\n                label = self.classes_dict[annotation[\"type\"]]\n                if label in self.classes:\n                    if coordinates:\n                        if self.normalize:\n                            coordinates = np.array(coordinates) / 512.0\n                        coordinates = \" \".join(map(str, chain(*coordinates)))\n                        file.write(f\"{label} {coordinates}\\n\")\n                        self.labels_counter += 1\n\n    def __splitfolders(self):\n        for i, line in tqdm(\n                enumerate(self.samples),\n                desc=\"Dataset creation\", total=self.length\n        ):\n            self.labels_counter = 0\n            identifier = line[\"id\"]\n            annotations = line[\"annotations\"]\n            paths_dict = self.__define_paths(i)\n\n            dst_image_path = self.__get_image_path(paths_dict, identifier)\n            dst_label_path = self.__get_label_path(paths_dict, identifier)\n\n            self.__copy_image(dst_image_path, identifier)\n            self.__copy_label(annotations, dst_label_path)\n\n            if self.labels_counter == 0:\n                os.remove(dst_image_path)\n                os.remove(dst_label_path)\n\n    def __count_dataset(self) -> dict:\n        train_images = len(os.listdir(os.path.join(self.train_dirpath, \"images\")))\n        train_labels = len(os.listdir(os.path.join(self.train_dirpath, \"labels\")))\n        val_images = len(os.listdir(os.path.join(self.val_dirpath, \"images\")))\n        val_labels = len(os.listdir(os.path.join(self.val_dirpath, \"labels\")))\n        return {\n            \"train_images\": train_images,\n            \"train_labels\": train_labels,\n            \"val_images\": val_images,\n            \"val_labels\": val_labels\n        }\n\n    @staticmethod\n    def __check_sanity(count_dict: dict) -> None:\n        assert count_dict[\"train_images\"] == count_dict[\"train_labels\"]\n        assert count_dict[\"val_images\"] == count_dict[\"val_labels\"]\n\n    def __finalizing(self, count_dict: dict) -> None:\n        assert os.path.exists(self.dataset_dirpath)\n\n        example_structure = [\n            \"dataset\",\n            \"train\", \"labels\", \"images\",\n            \"val\", \"labels\", \"images\"\n        ]\n\n        dir_bone = (\n            dirname.split(\"/\")[-1]\n            for dirname, _, filenames in os.walk(self.dataset_dirpath)\n            if dirname.split(\"/\")[-1] in example_structure\n        )\n\n        try:\n            print(\"\\n~ HuBMAP Dataset Structure ~\\n\")\n            print(\n            f\"\"\"\n            ├── {next(dir_bone)}\n            │   │\n            │   ├── {next(dir_bone)}\n            │   │   └── {next(dir_bone)}\n            │   │   └── {next(dir_bone)}\n            │   │\n            │   ├── {next(dir_bone)}\n            │   │   └── {next(dir_bone)}\n            │   │   └── {next(dir_bone)}\n            \"\"\"\n            )\n        except StopIteration as e:\n            print(e)\n        else:\n            print(Fore.GREEN + \"-> Success\")\n            print(Fore.GREEN + f\"Train dataset: {count_dict['train_images']}\\nVal dataset: {count_dict['val_images']}\")\n\n    def get_config(self) ->dict:\n        names = [\"blood_vessel\", \"glomerulus\", \"unsure\"]\n        return {\n            \"train\": str(self.train_dirpath),\n            \"val\": str(self.val_dirpath),\n            \"names\": [names[i] for i in self.classes]\n        }\n\n    @staticmethod\n    def display_config(config: dict) -> None:\n        print(Fore.BLACK + \"\\n~ HuBMAP Config Structure ~\\n\")\n        print(\n        f\"\"\"\n        │   │\n        │   ├── train\n        │   │   └── {config['train']}/images\n        │   │\n        │   │\n        │   ├── val\n        │   │   └── {config['val']}/images\n        │   │\n        │   │\n        │   ├── names\n        │   │   └── {' '.join(config['names'])}\n        \"\"\"\n        )\n        print(Fore.GREEN + \"-> Success\")\n        print(Fore.GREEN + f\"Number of classes: {len(config['names'])}\"\n                        f\"\\nClasses: {' '.join(config['names'])}\" \n            )\n\n    def write_config(self, config: dict) -> None:\n        with open(self.config_path, mode=\"w\") as f:\n            yaml.safe_dump(stream=f, data=config)\n\n    def __call__(self, train_size: float,\n                classes,\n                make_config: bool = True,\n                normalize: bool = True\n            ) -> None:\n        \n        self.train_size = train_size\n        self.classes = classes\n        self.normalize = normalize\n        \n        self.__define_splitratio()\n        self.__prepare_dirs()\n        self.__splitfolders()\n        count_dict = self.__count_dataset()\n        self.__check_sanity(count_dict)\n        self.__finalizing(count_dict)\n        \n        if make_config:\n            config = self.get_config()\n            self.write_config(config)\n            self.display_config(config)  \n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:21.784441Z","iopub.status.idle":"2023-12-10T13:58:21.784803Z","shell.execute_reply.started":"2023-12-10T13:58:21.784611Z","shell.execute_reply":"2023-12-10T13:58:21.784628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" dataset = Dataset(\nannotations_filepath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\",\nimages_dirpath=\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\",\n)\ndataset(train_size=0.80, classes=[0, 1, 2])","metadata":{"execution":{"iopub.status.busy":"2023-12-10T13:58:35.070872Z","iopub.execute_input":"2023-12-10T13:58:35.071597Z","iopub.status.idle":"2023-12-10T13:59:13.443524Z","shell.execute_reply.started":"2023-12-10T13:58:35.071555Z","shell.execute_reply":"2023-12-10T13:59:13.441999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install ultralytics\n","metadata":{"execution":{"iopub.status.busy":"2023-12-10T14:00:46.587903Z","iopub.execute_input":"2023-12-10T14:00:46.588735Z","iopub.status.idle":"2023-12-10T14:00:46.593196Z","shell.execute_reply.started":"2023-12-10T14:00:46.588704Z","shell.execute_reply":"2023-12-10T14:00:46.592312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ultralytics\nultralytics.checks()\nimport matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-12-10T14:00:40.879504Z","iopub.execute_input":"2023-12-10T14:00:40.879900Z","iopub.status.idle":"2023-12-10T14:00:46.586454Z","shell.execute_reply.started":"2023-12-10T14:00:40.879837Z","shell.execute_reply":"2023-12-10T14:00:46.585581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load a model\nmodel = YOLO('yolov8x-seg.pt') ","metadata":{"execution":{"iopub.status.busy":"2023-12-10T14:00:56.335724Z","iopub.execute_input":"2023-12-10T14:00:56.336215Z","iopub.status.idle":"2023-12-10T14:00:57.565317Z","shell.execute_reply.started":"2023-12-10T14:00:56.336172Z","shell.execute_reply":"2023-12-10T14:00:57.564499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.train(data='/kaggle/working/dataset/config.yaml',\n                    save=True,\n                    save_period=3,\n                    imgsz=512,\n                    project=\"HuBMAP\",\n                    name=\"yolov8x-seg\",\n                    deterministic=True,\n                    seed=42,\n                    device=0,\n                    val=True,\n                    workers=16,\n                    batch=16,\n                    epochs=50,\n                    lr0=1e-4,\n                    lrf=1e-2,\n                    cos_lr=True,\n                    momentum=0.937,\n                    weight_decay=0.0005,\n                    close_mosaic=10, \n                    optimizer=\"AdamW\",\n                    hsv_h= 0.015,\n                    \n                    \n                    hsv_s= 0.7,\n                    hsv_v= 0.4,\n                    degrees= 45.0,\n                    translate= 0.1,\n                    scale= 0.5,\n                    shear= 15.0,\n                    perspective= 0.0,\n                    flipud= 0.5,\n                    fliplr= 0.5,\n                    mosaic= 1.0,\n                    mixup= 1.0/3,\n                    copy_paste= 1.0/3,\n\n                    mask_ratio=1\n                    )","metadata":{"execution":{"iopub.status.busy":"2023-12-10T14:01:14.355038Z","iopub.execute_input":"2023-12-10T14:01:14.355433Z"},"trusted":true},"execution_count":null,"outputs":[]}]}