{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9187072,"sourceType":"datasetVersion","datasetId":5504483},{"sourceId":9195731,"sourceType":"datasetVersion","datasetId":5559249}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport pydicom\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport cv2\nimport glob\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-26T17:20:17.787405Z","iopub.execute_input":"2024-09-26T17:20:17.787806Z","iopub.status.idle":"2024-09-26T17:20:19.628563Z","shell.execute_reply.started":"2024-09-26T17:20:17.787775Z","shell.execute_reply":"2024-09-26T17:20:19.627174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf /kaggle/working/yolo_pretrain_data","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:20:19.630652Z","iopub.execute_input":"2024-09-26T17:20:19.631647Z","iopub.status.idle":"2024-09-26T17:20:20.703591Z","shell.execute_reply.started":"2024-09-26T17:20:19.631602Z","shell.execute_reply":"2024-09-26T17:20:20.701565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/lumbar-coordinate-pretraining-dataset/coords_pretrain.csv')\ndf['file_dir'] = df.apply(lambda x: f'/kaggle/input/lumbar-coordinate-pretraining-dataset/data/processed_{x.source}_jpgs/{x.filename}', axis = 1)\nprint('Generating data dir')\nOUT_DIR = f'yolo_pretrain_data'\nos.makedirs(OUT_DIR, exist_ok=True)\nphases = ['train', 'val']\n\nfor phase in phases:\n    img_dir = os.path.join(OUT_DIR, 'images', phase)\n    ann_dir = os.path.join(OUT_DIR, 'labels', phase)\n    os.makedirs(img_dir, exist_ok=True)\n    os.makedirs(ann_dir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:22:17.528790Z","iopub.execute_input":"2024-09-26T17:22:17.529709Z","iopub.status.idle":"2024-09-26T17:22:17.657031Z","shell.execute_reply.started":"2024-09-26T17:22:17.529659Z","shell.execute_reply":"2024-09-26T17:22:17.655874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df.source.isin(['spider', 'lsd']) ]","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:24:52.316990Z","iopub.execute_input":"2024-09-26T17:24:52.317365Z","iopub.status.idle":"2024-09-26T17:24:52.324935Z","shell.execute_reply.started":"2024-09-26T17:24:52.317338Z","shell.execute_reply":"2024-09-26T17:24:52.323789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OD_INPUT_SIZE = 384\nSTD_BOX_SIZE = 20\n\ndf_iter = df[['filename', 'source', 'file_dir']].drop_duplicates()\ndf_iter['source_code'] = df_iter.source.apply(lambda x : x[:3])\nlevel_dict = {k:v for v,k in enumerate(np.sort(df.level.unique()))}\n\n\n# Training data creation\nfor idx, row in df_iter.iterrows():\n    H, W = 256, 256\n    source_file = row['file_dir']\n    filename = row['source_code'] + row['filename']\n    destination_dir = '/kaggle/working/yolo_pretrain_data/images/train'\n    destination_file = os.path.join(destination_dir, filename)\n    shutil.copy(source_file, destination_file)\n        \n    ann_dir = '/kaggle/working/yolo_pretrain_data/labels/train'\n    ann_path = os.path.join(ann_dir, filename[:-4] + '.txt')\n    filter_df = df[(df['source'] == row.source) & (df['filename'] == row.filename) ]\n    with open(ann_path, 'w') as f:\n        for i, row in filter_df.iterrows():\n#             cond = row['condition']\n            level = row['level']\n            class_id = level_dict[level]\n            x_center = row['x'] / W\n            y_center = row['y'] / H\n            width = W / OD_INPUT_SIZE * STD_BOX_SIZE / W\n            height = H /  OD_INPUT_SIZE * STD_BOX_SIZE / H\n            f.write(f'{class_id} {x_center} {y_center} {width} {height}\\n')\n            \n            \n# Validation folder creation\ndf_iter_val = df_iter.sample(100, random_state = 100)\n\nfor idx, row in df_iter_val.iterrows():\n    H, W = 256, 256\n    source_file = row['file_dir']\n    filename = row['source_code'] + row['filename']\n    destination_dir = '/kaggle/working/yolo_pretrain_data/images/val'\n    destination_file = os.path.join(destination_dir, filename)\n    shutil.copy(source_file, destination_file)\n        \n    ann_dir = '/kaggle/working/yolo_pretrain_data/labels/val'\n    ann_path = os.path.join(ann_dir, filename[:-4] + '.txt')\n    filter_df = df[(df['source'] == row.source) & (df['filename'] == row.filename) ]\n    with open(ann_path, 'w') as f:\n        for i, row in filter_df.iterrows():\n#             cond = row['condition']\n            level = row['level']\n            class_id = level_dict[level]\n            x_center = row['x'] / W\n            y_center = row['y'] / H\n            width = W / OD_INPUT_SIZE * STD_BOX_SIZE / W\n            height = H /  OD_INPUT_SIZE * STD_BOX_SIZE / H\n            f.write(f'{class_id} {x_center} {y_center} {width} {height}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:24:58.682726Z","iopub.execute_input":"2024-09-26T17:24:58.683108Z","iopub.status.idle":"2024-09-26T17:25:04.346747Z","shell.execute_reply.started":"2024-09-26T17:24:58.683077Z","shell.execute_reply":"2024-09-26T17:25:04.345581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # test of train\n\n_IM_DIR = f'{OUT_DIR}/images/train'\n_ANN_DIR = f'{OUT_DIR}/labels/train'\nname = np.random.choice(os.listdir(_IM_DIR))[:-4]\n\n\nprint(name)\n\nim = plt.imread(os.path.join(_IM_DIR, name+'.jpg')).copy()\nH,W = im.shape[:2]\nanns = np.loadtxt(os.path.join(_ANN_DIR, name+'.txt')).reshape(-1, 5)\n\nfor _cls, x,y,w,h in anns.tolist():\n    x *= W\n    y *= H\n    w *= W\n    h *= H\n    x1 = int(x-w/2)\n    x2 = int(x+w/2)\n    y1 = int(y-h/2)\n    y2 = int(y+h/2)\n    label = _cls\n    \n    c = (0,255,255)\n\n    im = cv2.rectangle(im, (x1,y1), (x2,y2), c, 2)\n#     cv2.putText(im, label, (x1,y1), fontFace, 0.3, c, 1, cv2.LINE_AA)\n\n\nplt.imshow(im)\nprint (anns)\ndf[(df['filename'] == name[3:] + '.jpg')]","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:25:08.413690Z","iopub.execute_input":"2024-09-26T17:25:08.414046Z","iopub.status.idle":"2024-09-26T17:25:08.809940Z","shell.execute_reply.started":"2024-09-26T17:25:08.414020Z","shell.execute_reply":"2024-09-26T17:25:08.808785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # test of val\n\n_IM_DIR = f'{OUT_DIR}/images/val'\n_ANN_DIR = f'{OUT_DIR}/labels/val'\nname = np.random.choice(os.listdir(_IM_DIR))[:-4]\n\n\nprint(name)\n\nim = plt.imread(os.path.join(_IM_DIR, name+'.jpg')).copy()\nH,W = im.shape[:2]\nanns = np.loadtxt(os.path.join(_ANN_DIR, name+'.txt')).reshape(-1, 5)\n\nfor _cls, x,y,w,h in anns.tolist():\n    x *= W\n    y *= H\n    w *= W\n    h *= H\n    x1 = int(x-w/2)\n    x2 = int(x+w/2)\n    y1 = int(y-h/2)\n    y2 = int(y+h/2)\n    label = _cls\n    \n    c = (0,255,255)\n\n    im = cv2.rectangle(im, (x1,y1), (x2,y2), c, 2)\n#     cv2.putText(im, label, (x1,y1), fontFace, 0.3, c, 1, cv2.LINE_AA)\n\n\nplt.imshow(im)\nprint (anns)\ndf[(df['filename'] == name[3:] + '.jpg')]","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:25:11.502711Z","iopub.execute_input":"2024-09-26T17:25:11.503089Z","iopub.status.idle":"2024-09-26T17:25:12.162063Z","shell.execute_reply.started":"2024-09-26T17:25:11.503061Z","shell.execute_reply":"2024-09-26T17:25:12.160845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r -q yolo_pretrain_data.zip yolo_pretrain_data\n!rm -rf yolo_pretrain_data","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:25:14.857120Z","iopub.execute_input":"2024-09-26T17:25:14.857514Z","iopub.status.idle":"2024-09-26T17:25:17.844399Z","shell.execute_reply.started":"2024-09-26T17:25:14.857478Z","shell.execute_reply":"2024-09-26T17:25:17.842923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(level_dict)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T17:25:17.846536Z","iopub.execute_input":"2024-09-26T17:25:17.846875Z","iopub.status.idle":"2024-09-26T17:25:17.853179Z","shell.execute_reply.started":"2024-09-26T17:25:17.846845Z","shell.execute_reply":"2024-09-26T17:25:17.851905Z"},"trusted":true},"execution_count":null,"outputs":[]}]}