{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9195731,"sourceType":"datasetVersion","datasetId":5559249},{"sourceId":193161758,"sourceType":"kernelVersion"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport pydicom\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport cv2\nimport glob","metadata":{"_uuid":"764c284e-5e10-4726-a430-4f7e215a7993","_cell_guid":"52d318ba-52cf-4f6a-abc2-cfa79746df42","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:56:03.697575Z","iopub.execute_input":"2024-09-15T10:56:03.698345Z","iopub.status.idle":"2024-09-15T10:56:04.674019Z","shell.execute_reply.started":"2024-09-15T10:56:03.698303Z","shell.execute_reply":"2024-09-15T10:56:04.672809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_DIR = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images\"","metadata":{"_uuid":"d7672421-7ded-4017-a242-f3aa3e77f8ba","_cell_guid":"4d5127a2-9d2e-4f06-8260-770ff56e9b50","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:56:04.805784Z","iopub.execute_input":"2024-09-15T10:56:04.806873Z","iopub.status.idle":"2024-09-15T10:56:04.811261Z","shell.execute_reply.started":"2024-09-15T10:56:04.806835Z","shell.execute_reply":"2024-09-15T10:56:04.810166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = [0,1,2,3,4]\nOD_INPUT_SIZE = 384\nSTD_BOX_SIZE = 20\nSAMPLE = 10\nCONDITIONS = ['Spinal Canal Stenosis']\nSEVERITIES = ['Normal/Mild', 'Moderate', 'Severe']\nLEVELS = ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']","metadata":{"_uuid":"1be3024e-5019-407f-be4b-3ebf48fba1b1","_cell_guid":"2b2c4acb-49c8-42d5-85e5-c6b663480f84","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:23.375601Z","iopub.execute_input":"2024-09-15T10:58:23.376033Z","iopub.status.idle":"2024-09-15T10:58:23.382081Z","shell.execute_reply.started":"2024-09-15T10:58:23.376000Z","shell.execute_reply":"2024-09-15T10:58:23.380626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -rf val_fold0","metadata":{"_uuid":"978abdb7-d359-4147-b31f-7085f662c5fd","_cell_guid":"3af1f3a4-76da-466f-9f47-ec80c3ebd5ac","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:02.283972Z","iopub.execute_input":"2024-09-15T10:58:02.284353Z","iopub.status.idle":"2024-09-15T10:58:02.288923Z","shell.execute_reply.started":"2024-09-15T10:58:02.284323Z","shell.execute_reply":"2024-09-15T10:58:02.287804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_val_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ntrain_xy = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ntrain_des = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')","metadata":{"_uuid":"1d547b44-ee84-4b37-830a-a453a6463e92","_cell_guid":"02039940-e8df-4da3-b4f5-fa5603ec6e74","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:24.319828Z","iopub.execute_input":"2024-09-15T10:58:24.320214Z","iopub.status.idle":"2024-09-15T10:58:24.412202Z","shell.execute_reply.started":"2024-09-15T10:58:24.320185Z","shell.execute_reply":"2024-09-15T10:58:24.411175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SAMPLE:\n    train_val_df = train_val_df.sample(SAMPLE, random_state=2698)","metadata":{"_uuid":"89535e60-c3f3-4581-a848-dd7b2b5ee712","_cell_guid":"7ac867c4-aec7-4126-bcef-2be2aa2d4718","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:24.994957Z","iopub.execute_input":"2024-09-15T10:58:24.995360Z","iopub.status.idle":"2024-09-15T10:58:25.006566Z","shell.execute_reply.started":"2024-09-15T10:58:24.995328Z","shell.execute_reply":"2024-09-15T10:58:25.005250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_df = pd.read_csv('/kaggle/input/lsdc-fold-split/5folds.csv')","metadata":{"_uuid":"07a1582c-7c83-413d-937b-05e2cb2f18ff","_cell_guid":"f640a598-2ea7-4be3-90ed-eea444d0a1a9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:25.865013Z","iopub.execute_input":"2024-09-15T10:58:25.865758Z","iopub.status.idle":"2024-09-15T10:58:25.873504Z","shell.execute_reply.started":"2024-09-15T10:58:25.865708Z","shell.execute_reply":"2024-09-15T10:58:25.872459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_xy.head(3)","metadata":{"_uuid":"e8b8fa2a-4839-461b-9dc2-eca19909d12d","_cell_guid":"b48bddde-6f79-40a8-a526-e2f611f59e91","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:26.639316Z","iopub.execute_input":"2024-09-15T10:58:26.639700Z","iopub.status.idle":"2024-09-15T10:58:26.652238Z","shell.execute_reply.started":"2024-09-15T10:58:26.639669Z","shell.execute_reply":"2024-09-15T10:58:26.651128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_level(text):\n    # 遍历所有的椎间盘等级\n    for lev in ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']:\n        # 如果找到与输入文本匹配的等级\n        if lev in text:\n            split = lev.split('_')  # 通过下划线分割等级\n            split[0] = split[0].capitalize()  # 首字母大写\n            split[1] = split[1].capitalize()  # 首字母大写\n            return '/'.join(split)  # 使用斜杠连接两个部分\n    # 如果没有找到匹配的等级，则抛出异常\n    raise ValueError('Level not found ' + lev)\n    \ndef get_condition(text):\n    # 通过下划线将文本分割为多个部分\n    split = text.split('_')\n    # 将每个部分的首字母大写\n    for i in range(len(split)):\n        split[i] = split[i].capitalize()\n    # 去掉最后两部分，因为它们是椎间盘等级信息\n    split = split[:-2]\n    # 将剩余的部分通过空格连接，并返回\n    return ' '.join(split)\n#     raise ValueError('Condition not found '+ lev)  # 注释掉的异常处理部分\n","metadata":{"_uuid":"7f43a3da-fd3a-4cc0-aaa4-50355120ca89","_cell_guid":"352b0e37-8bbd-4354-8dc9-2e2bd723df19","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:27.964753Z","iopub.execute_input":"2024-09-15T10:58:27.965154Z","iopub.status.idle":"2024-09-15T10:58:27.973231Z","shell.execute_reply.started":"2024-09-15T10:58:27.965124Z","shell.execute_reply":"2024-09-15T10:58:27.971848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_xy['condition'].unique()","metadata":{"_uuid":"8adc9d37-2406-463f-9c70-35883f180ba7","_cell_guid":"69a6f3e6-9edc-42a1-8e26-1f86c778e2b3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:29.245655Z","iopub.execute_input":"2024-09-15T10:58:29.246080Z","iopub.status.idle":"2024-09-15T10:58:29.257232Z","shell.execute_reply.started":"2024-09-15T10:58:29.246047Z","shell.execute_reply":"2024-09-15T10:58:29.255921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = train_df.dropna()","metadata":{"_uuid":"5cdf2c80-69d9-4a31-a06e-b018cfc48c69","_cell_guid":"b92a0873-5a06-437c-ad73-46a61661213d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T10:58:29.964468Z","iopub.execute_input":"2024-09-15T10:58:29.964871Z","iopub.status.idle":"2024-09-15T10:58:29.969365Z","shell.execute_reply.started":"2024-09-15T10:58:29.964838Z","shell.execute_reply":"2024-09-15T10:58:29.968141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_level('spinal_canal_stenosis_l1_l2')","metadata":{"execution":{"iopub.status.busy":"2024-09-15T11:00:42.393694Z","iopub.execute_input":"2024-09-15T11:00:42.394108Z","iopub.status.idle":"2024-09-15T11:00:42.400711Z","shell.execute_reply.started":"2024-09-15T11:00:42.394081Z","shell.execute_reply":"2024-09-15T11:00:42.399656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_condition('spinal_canal_stenosis_l1_l2')","metadata":{"execution":{"iopub.status.busy":"2024-09-15T11:01:34.751928Z","iopub.execute_input":"2024-09-15T11:01:34.752318Z","iopub.status.idle":"2024-09-15T11:01:34.759156Z","shell.execute_reply.started":"2024-09-15T11:01:34.752286Z","shell.execute_reply":"2024-09-15T11:01:34.757949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## debug","metadata":{}},{"cell_type":"code","source":"train_val_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-15T11:02:40.228667Z","iopub.execute_input":"2024-09-15T11:02:40.229732Z","iopub.status.idle":"2024-09-15T11:02:40.255425Z","shell.execute_reply.started":"2024-09-15T11:02:40.229687Z","shell.execute_reply":"2024-09-15T11:02:40.254124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 创建一个字典用于存储标签数据\nlabel_df = {'study_id':[], 'condition': [], 'level':[], 'label':[]}\n\n# 遍历训练/验证数据框中的每一行\nfor i, row in train_val_df.iterrows():\n    study_id = row['study_id']  # 获取当前行的study_id\n    print('study_id',study_id)\n    # 遍历当前行中除study_id外的所有标签\n\n    for k, label in row.iloc[1:].to_dict().items():\n        print('k:',k)#'spinal_canal_stenosis_l1_l2'\n        level = get_level(k)  # 从标签键中提取椎间盘等级'L1/L2'\n        condition = get_condition(k)  # 从标签键中提取病情'Spinal Canal Stenosis'\n        print('LABEL:',label)\n        # 将信息添加到label_df字典中\n        label_df['study_id'].append(study_id)\n        label_df['condition'].append(condition)\n        label_df['level'].append(level)\n        label_df['label'].append(label)\n#         break  # 调试时用于只处理一个标签，注释掉\n#     break  # 调试时用于只处理一行，注释掉\n\n# 将字典转换为DataFrame\nlabel_df = pd.DataFrame(label_df)\n# 将label_df与fold_df按study_id进行合并\nlabel_df = label_df.merge(fold_df, on='study_id')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-15T11:02:12.180075Z","iopub.execute_input":"2024-09-15T11:02:12.180434Z","iopub.status.idle":"2024-09-15T11:02:12.197389Z","shell.execute_reply.started":"2024-09-15T11:02:12.180408Z","shell.execute_reply":"2024-09-15T11:02:12.196301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 创建一个字典用于存储标签数据\nlabel_df = {'study_id':[], 'condition': [], 'level':[], 'label':[]}\n\n# 遍历训练/验证数据框中的每一行\nfor i, row in train_val_df.iterrows():\n    study_id = row['study_id']  # 获取当前行的study_id\n    # 遍历当前行中除study_id外的所有标签\n    for k, label in row.iloc[1:].to_dict().items():\n        level = get_level(k)  # 从标签键中提取椎间盘等级\n        condition = get_condition(k)  # 从标签键中提取病情\n        # 将信息添加到label_df字典中\n        label_df['study_id'].append(study_id)\n        label_df['condition'].append(condition)\n        label_df['level'].append(level)\n        label_df['label'].append(label)\n#         break  # 调试时用于只处理一个标签，注释掉\n#     break  # 调试时用于只处理一行，注释掉\n\n# 将字典转换为DataFrame\nlabel_df = pd.DataFrame(label_df)\n# 将label_df与fold_df按study_id进行合并\nlabel_df = label_df.merge(fold_df, on='study_id')\n","metadata":{"_uuid":"bd5c47e0-9cd4-474b-b009-1c21a898307d","_cell_guid":"b06ab6d7-036f-4ff8-be73-237f096d40c0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-09-15T09:00:50.672512Z","iopub.execute_input":"2024-09-15T09:00:50.672934Z","iopub.status.idle":"2024-09-15T09:00:51.376431Z","shell.execute_reply.started":"2024-09-15T09:00:50.672899Z","shell.execute_reply":"2024-09-15T09:00:51.375370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_xy = train_xy.merge(train_des, how='inner', on=['study_id', 'series_id'])\nlabel_df = label_df.merge(train_xy, how='inner', on=['study_id', 'condition', 'level'])","metadata":{"_uuid":"65d607dd-bda2-4561-baa9-d9373062a82c","_cell_guid":"7b9b884c-0ed2-43ce-9f86-b8df9f179964","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_df[label_df.series_id.isna()]","metadata":{"_uuid":"5b0d7326-fbdb-4043-8be3-4313952545f6","_cell_guid":"5772819e-630f-4f38-90fc-2c10ffd8f85e","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnt = train_xy.groupby(['study_id', 'series_id', 'instance_number'])['condition'].nunique()","metadata":{"_uuid":"59714a9f-a20a-46b7-8825-6f913ae69d9a","_cell_guid":"9eb1af59-e8d0-4c52-8bff-b70903cd3741","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnt[cnt>1]","metadata":{"_uuid":"02a0425d-0b6b-43b9-a8ac-a30169e66e5f","_cell_guid":"a47bb1fe-25c8-4691-999f-b86f11daa2dc","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def query_train_xy_row(study_id, series_id=None, instance_num=None):\n    if series_id is not None and instance_num is not None:\n        return label_df[(label_df.study_id==study_id) & (label_df.series_id==series_id) &\n            (label_df.instance_number==instance_num)]\n    elif series_id is None and instance_num is None:\n        return label_df[(label_df.study_id==study_id)]\n    else:\n        return label_df[(train_xy.study_id==study_id) & (label_df.series_id==series_id)]","metadata":{"_uuid":"f1eb6587-7f2c-479e-ad0b-e8a13495014a","_cell_guid":"ccb84dc5-e47c-4914-928f-ce500298c855","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n\n# def count_dcm_files(directory):\n#     dcm_count = 0\n#     for root, dirs, files in os.walk(directory):\n#         for file in files:\n#             if file.endswith('.dcm'):\n#                 dcm_count += 1\n#     return dcm_count\n\n# dcm_files_count = count_dcm_files(IMG_DIR)\n\n# print(f\"Number of .dcm files: {dcm_files_count}\")","metadata":{"_uuid":"5c3176f7-a2c1-4a3e-9d92-aa00be46ba98","_cell_guid":"f2236a21-f126-4756-9679-5e10e82dd0a0","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_dcm(src_path):\n    dicom_data = pydicom.dcmread(src_path)\n    image = dicom_data.pixel_array\n    image = (image - image.min()) / (image.max() - image.min() +1e-6) * 255\n    image = np.stack([image]*3, axis=-1).astype('uint8')\n    return image\n\ndef get_accronym(text):\n    split = text.split(' ')\n    return ''.join([x[0] for x in split])","metadata":{"_uuid":"561d8c40-b61b-413e-80cd-395c308fea3a","_cell_guid":"93be66eb-34d9-4b38-a234-6a5db8ddec5f","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# study_id = 4003253 \n# series_id = 2448190387\n# instance_num = 28\n\nex = label_df.sample(1).iloc[0]\nstudy_id = ex.study_id\nseries_id = ex.series_id\ninstance_num = ex.instance_number\n\nWIDTH = 10\n\npath = os.path.join(IMG_DIR, str(study_id), str(series_id), f'{instance_num}.dcm')","metadata":{"_uuid":"02916b67-2259-447a-9e9d-3cb3cb380875","_cell_guid":"e0844eb7-5046-46fd-859b-3318fe5a11b3","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = read_dcm(path)\n\ntmp_df = query_train_xy_row(study_id, series_id, instance_num)\nfor i, row in tmp_df.iterrows():\n    lbl = f\"{get_accronym(row['condition'])}_{row['level']}\"\n    x, y = row['x'], row['y']\n    x1 = int(x - WIDTH)\n    x2 = int(x + WIDTH)\n    y1 = int(y - WIDTH)\n    y2 = int(y + WIDTH)\n    color = None\n    if row['label'] == 'Normal/Mild':\n        color =  (0, 255, 0)\n    elif row['label'] == 'Moderate':\n        color = (255,255,0) \n    elif row['label'] == 'Severe':\n        color = (255,0,0)\n        \n    fontFace = cv2.FONT_HERSHEY_SIMPLEX\n    fontScale = 0.5\n    thickness = 1\n    cv2.rectangle(img, (x1,y1), (x2,y2), color, 2)\n    cv2.putText(img, lbl, (x1,y1), fontFace, fontScale, color, thickness, cv2.LINE_AA)\n\ntmp_df","metadata":{"_uuid":"0d552f09-d1b0-4876-8582-8a00684a9500","_cell_guid":"a42b449a-cfc3-446d-9b4e-7799fb2ee937","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img)\nplt.show()","metadata":{"_uuid":"2909dbbf-59d3-49a8-8002-44449d9a7864","_cell_guid":"2b799156-76c2-468d-9563-8ecbf782905d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_df[['study_id', 'series_id']].drop_duplicates()","metadata":{"_uuid":"40e4ee56-f7d5-43ef-b538-51910c2d3326","_cell_guid":"75dca154-c4d1-4884-9208-435cece6e799","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_dcm(src_path):\n    dicom_data = pydicom.dcmread(src_path)\n    image = dicom_data.pixel_array\n    image = (image - image.min()) / (image.max() - image.min() +1e-6) * 255\n    image = np.stack([image]*3, axis=-1).astype('uint8')\n    return image","metadata":{"_uuid":"ed329c2b-4862-4702-96f6-b0e60408380b","_cell_guid":"2b0b952d-e479-4415-a318-ed3e86496305","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_df = label_df[label_df.condition.map(lambda x: x in CONDITIONS)]","metadata":{"_uuid":"d6ee535e-966b-4b47-acf2-975f539e5c0e","_cell_guid":"f11ad3b3-8574-442d-9ba8-c94d15690612","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label2id = {}\nid2label = {}\ni = 0\nfor cond in CONDITIONS:\n    for level in LEVELS:\n        for severity in SEVERITIES:\n            cls_ = f\"{cond.lower().replace(' ', '_')}_{level}_{severity.lower()}\"\n            label2id[cls_] = i\n            id2label[i] = cls_\n            i+=1","metadata":{"_uuid":"c2a7cc7a-75b2-4657-9011-be30915f27f3","_cell_guid":"be5b9666-5fe5-42d3-b982-d013b070cb85","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id2label","metadata":{"_uuid":"33a16829-73cc-4ae8-9341-3e6bdc214d5a","_cell_guid":"4777e88b-22cb-409a-9913-c34cfc3f36b0","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gen_yolo_format(ann_df, phase='train'):\n    for name, group in tqdm(ann_df.groupby(['study_id', 'series_id', 'instance_number'])):\n        study_id, series_id, instance_num = name[0], name[1], name[2]\n        path = f'{IMG_DIR}/{study_id}/{series_id}/{instance_num}.dcm'\n        img = read_dcm(path)\n        H, W = img.shape[:2]\n\n        img_dir = os.path.join(OUT_DIR, 'images', phase)\n        os.makedirs(img_dir, exist_ok=True)\n        img_path = os.path.join(img_dir, f'{study_id}_{series_id}_{instance_num}.jpg')\n        cv2.imwrite(img_path, img)\n\n        ann_dir = os.path.join(OUT_DIR, 'labels', phase)\n        os.makedirs(ann_dir, exist_ok=True)\n        ann_path = os.path.join(ann_dir, f'{study_id}_{series_id}_{instance_num}.txt')\n        with open(ann_path, 'w') as f:\n            for i, row in group.iterrows():\n                cond = row['condition']\n                level = row['level']\n                severity = row['label']\n                class_label = f\"{cond.lower().replace(' ', '_')}_{level.lower().replace('/', '_')}_{severity.lower()}\"\n                class_id = label2id[class_label]\n                x_center = row['x'] / W\n                y_center = row['y'] / H\n                width = W / OD_INPUT_SIZE * STD_BOX_SIZE / W\n                height = H /  OD_INPUT_SIZE * STD_BOX_SIZE / H\n                f.write(f'{class_id} {x_center} {y_center} {width} {height}\\n')\n\n#         break","metadata":{"_uuid":"e81709f5-e0a4-47f3-9d37-2ceee9a91251","_cell_guid":"021428db-57a3-4648-9c3f-f6a64773fe20","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for FOLD in FOLDS:\n    print('Gen data fold', FOLD)\n    OUT_DIR = f'data_fold{FOLD}'\n    os.makedirs(OUT_DIR, exist_ok=True)\n    \n    train_df = filtered_df[filtered_df.fold != FOLD]\n    val_df = filtered_df[filtered_df.fold == FOLD]\n    \n    gen_yolo_format(train_df, phase='train')\n    gen_yolo_format(val_df, phase='val')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls data_fold0","metadata":{"_uuid":"ede352e2-3fe1-490a-bba8-052734665033","_cell_guid":"97e4af82-fe6b-4075-b4a9-d2249f0a6c00","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # test generated annotations\n\n_IM_DIR = f'{OUT_DIR}/images/train'\n_ANN_DIR = f'{OUT_DIR}/labels/train'\nname = np.random.choice(os.listdir(_IM_DIR))[:-4]\n\nim = plt.imread(os.path.join(_IM_DIR, name+'.jpg')).copy()\nH,W = im.shape[:2]\nanns = np.loadtxt(os.path.join(_ANN_DIR, name+'.txt')).reshape(-1, 5)\n\nfor _cls, x,y,w,h in anns.tolist():\n    x *= W\n    y *= H\n    w *= W\n    h *= H\n    x1 = int(x-w/2)\n    x2 = int(x+w/2)\n    y1 = int(y-h/2)\n    y2 = int(y+h/2)\n    label = id2label[_cls]\n    \n#     if _cls == 0:\n#         c = (255,0,0)\n#     elif _cls == 1:\n#         c = (0,255,0)\n#     else:\n#         c = (255,255,0)\n    c = (0,255,255)\n\n    im = cv2.rectangle(im, (x1,y1), (x2,y2), c, 2)\n    cv2.putText(im, label, (x1,y1), fontFace, 0.3, c, 1, cv2.LINE_AA)\n\n\nplt.imshow(im)","metadata":{"_uuid":"a4915576-5e6b-49c1-931c-6f6d37062e9f","_cell_guid":"1eb1b752-bcbf-48d5-9c08-c55b7cddb6f4","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k, v in id2label.items():\n    print(f'{k}: {v}')","metadata":{"_uuid":"718e84b2-024f-4542-981c-b26b458076c6","_cell_guid":"68a6ba36-026e-48d6-9614-965160b6ea1b","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r -q data_fold0.zip data_fold0\n!zip -r -q data_fold1.zip data_fold1\n!zip -r -q data_fold2.zip data_fold2\n!zip -r -q data_fold3.zip data_fold3\n!zip -r -q data_fold4.zip data_fold4","metadata":{"_uuid":"d092d8bc-ac0d-4e43-9de0-22125563ddc3","_cell_guid":"6f1c1c6f-1cfa-406f-9348-d88bf8f1a6db","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf data_fold0\n!rm -rf data_fold1\n!rm -rf data_fold2\n!rm -rf data_fold3\n!rm -rf data_fold4","metadata":{"_uuid":"d4edd23e-0d2b-4329-9a10-dd5ac3c7806d","_cell_guid":"8d76bb94-a645-45aa-bee9-436804d8c25a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"_uuid":"d33aad5d-5f59-418f-9e86-82281c92df1f","_cell_guid":"65c60638-64d7-448f-87db-d8551e65ec5c","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"ea16dad1-2c23-41a3-a091-fd4feeaded68","_cell_guid":"d39a9370-1c34-4ba1-9f35-b6d991f76d0b","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}