{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":6108,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":4651}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-11T11:05:47.768167Z","iopub.execute_input":"2024-02-11T11:05:47.768494Z","iopub.status.idle":"2024-02-11T11:05:48.902661Z","shell.execute_reply.started":"2024-02-11T11:05:47.768432Z","shell.execute_reply":"2024-02-11T11:05:48.901708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport yaml\n\nimport pydicom\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:05:48.904502Z","iopub.execute_input":"2024-02-11T11:05:48.905060Z","iopub.status.idle":"2024-02-11T11:05:49.184822Z","shell.execute_reply.started":"2024-02-11T11:05:48.905025Z","shell.execute_reply":"2024-02-11T11:05:49.183596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#From DICOM to jpg (train images)\n\ntrain_images_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\njpg_dir_path = '/kaggle/working/train_images'\nos.makedirs(jpg_dir_path, exist_ok=True)\n\ndef dicom_to_jpg(dicom_path, jpg_path):\n    dicom_data = pydicom.dcmread(dicom_path)\n    image_array = dicom_data.pixel_array\n    image = Image.fromarray(image_array)\n    image.save(jpg_path)\n\nfor filename in os.listdir(train_images_path):\n    dicom_path = os.path.join(train_images_path, filename)\n    jpg_filename = f\"{os.path.splitext(filename)[0]}.jpg\"\n    jpg_path = os.path.join(jpg_dir_path, jpg_filename)\n\n    dicom_to_jpg(dicom_path, jpg_path)\n  ","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:05:49.190776Z","iopub.execute_input":"2024-02-11T11:05:49.191496Z","iopub.status.idle":"2024-02-11T11:13:39.519223Z","shell.execute_reply.started":"2024-02-11T11:05:49.191463Z","shell.execute_reply":"2024-02-11T11:13:39.518465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#From DICOM to jpg (test images)\n\ntest_images_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images'\njpg_dir_path = '/kaggle/working/test_images'\nos.makedirs(jpg_dir_path, exist_ok=True)\n\ndef dicom_to_jpg(dicom_path, jpg_path):\n    dicom_data = pydicom.dcmread(dicom_path)\n    image_array = dicom_data.pixel_array\n    image = Image.fromarray(image_array)\n    image.save(jpg_path)\n\nfor filename in os.listdir(test_images_path):\n    dicom_path = os.path.join(test_images_path, filename)\n    jpg_filename = f\"{os.path.splitext(filename)[0]}.jpg\"\n    jpg_path = os.path.join(jpg_dir_path, jpg_filename)\n\n    dicom_to_jpg(dicom_path, jpg_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:13:39.520854Z","iopub.execute_input":"2024-02-11T11:13:39.521142Z","iopub.status.idle":"2024-02-11T11:14:29.024677Z","shell.execute_reply.started":"2024-02-11T11:13:39.521116Z","shell.execute_reply":"2024-02-11T11:14:29.023894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imgs_dir=\"/kaggle/working/train_images\"\ntrain_labels=\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\n\ntest_imgs_dir=\"/kaggle/working/test_images\"\n\nimgs_list=list(sorted(os.listdir(train_imgs_dir)))\nidxs=list(range(len(imgs_list)))\nnp.random.shuffle(idxs)\n\ntrain_idx=idxs[:int(0.8*len(idxs))]\nval_idx=idxs[int(0.8*len(idxs)):]","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:29.025607Z","iopub.execute_input":"2024-02-11T11:14:29.025877Z","iopub.status.idle":"2024-02-11T11:14:29.060379Z","shell.execute_reply.started":"2024-02-11T11:14:29.025853Z","shell.execute_reply":"2024-02-11T11:14:29.059683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#yolo dataset format \n\n# root directory\n!mkdir \"/kaggle/working/data\"\n\n# images directory \n!mkdir \"/kaggle/working/data/images\"\n\n# train and test subdirectories with image directory\n!mkdir \"/kaggle/working/data/images/train\"\n!mkdir \"/kaggle/working/data/images/val\"\n\n# labels directory\n!mkdir \"/kaggle/working/data/labels\"\n\n# train and test subdirectories with labels directory\n!mkdir \"/kaggle/working/data/labels/train\"\n!mkdir \"/kaggle/working/data/labels/val\"\n\nroot_dir=\"/kaggle/working/data\"\nlabels_dir=\"/kaggle/working/data/labels\"\nimages_dir=\"/kaggle/working/data/images\"","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:29.061361Z","iopub.execute_input":"2024-02-11T11:14:29.061602Z","iopub.status.idle":"2024-02-11T11:14:35.696758Z","shell.execute_reply.started":"2024-02-11T11:14:29.061579Z","shell.execute_reply":"2024-02-11T11:14:35.695414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\n\ndata = pd.read_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:35.698386Z","iopub.execute_input":"2024-02-11T11:14:35.698722Z","iopub.status.idle":"2024-02-11T11:14:35.797438Z","shell.execute_reply.started":"2024-02-11T11:14:35.698691Z","shell.execute_reply":"2024-02-11T11:14:35.796672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"width=528\nheight=942\n\ntarget_data = data[data['Target'] == 1]\nzero_target_data = data[data['Target'] == 0]\nzero_target_data = zero_target_data.rename(columns = {'x':'x_centre', 'y':'y_centre'})\n\ntarget_data[\"x_centre\"] = (target_data[\"x\"] + target_data[\"width\"]) / 2\ntarget_data[\"y_centre\"] = (target_data[\"y\"] + target_data[\"height\"]) / 2\n\nmax_x_centre = max(target_data[\"x_centre\"])\nmax_y_centre = max(target_data[\"y_centre\"])\n\ntarget_data[\"x_centre\"] = target_data[\"x_centre\"] / max_x_centre\ntarget_data[\"y_centre\"] = target_data[\"y_centre\"] / max_y_centre\n\ntarget_data[\"width\"] /= width\ntarget_data[\"height\"] /= height\n\nnormalized_data=target_data[[\"patientId\",\"x_centre\",\"y_centre\",\"width\",\"height\", \"Target\"]]\nnormalized_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:35.798581Z","iopub.execute_input":"2024-02-11T11:14:35.798929Z","iopub.status.idle":"2024-02-11T11:14:35.861140Z","shell.execute_reply.started":"2024-02-11T11:14:35.798901Z","shell.execute_reply":"2024-02-11T11:14:35.860281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_yolo = pd.concat([normalized_data, zero_target_data])\n\ndf_yolo.describe()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:35.863962Z","iopub.execute_input":"2024-02-11T11:14:35.864222Z","iopub.status.idle":"2024-02-11T11:14:35.904474Z","shell.execute_reply.started":"2024-02-11T11:14:35.864199Z","shell.execute_reply":"2024-02-11T11:14:35.903593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_yolo.to_csv('train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:35.905900Z","iopub.execute_input":"2024-02-11T11:14:35.906212Z","iopub.status.idle":"2024-02-11T11:14:36.099499Z","shell.execute_reply.started":"2024-02-11T11:14:35.906182Z","shell.execute_reply":"2024-02-11T11:14:36.098782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx,img_name in enumerate(imgs_list):\n    subset=\"train\"\n    if idx in val_idx:\n        subset=\"val\"\n        \n    if np.isin(img_name,df_yolo[\"patientId\"]):\n        columns=[\"patientId\",\"x_centre\",\"y_centre\",\"width\",\"height\", \"Target\"]\n        img_bbox=df_yolo[df_yolo[\"img_name\"]==img_name][columns].values\n        \n        label_file_path=os.path.join(labels_dir,subset,img_name[:-4]+\".txt\")\n        with open(label_file_path,\"w+\") as f:\n            for row in img_bbox:\n                text=\" \".join(row.astype(str))\n                f.write(text)\n                f.write(\"\\n\")\n                \n    old_image_path=os.path.join(train_imgs_dir,img_name)\n    new_image_path=os.path.join(images_dir,subset,img_name)\n    shutil.copy(old_image_path,new_image_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:14:36.100560Z","iopub.execute_input":"2024-02-11T11:14:36.100855Z","iopub.status.idle":"2024-02-11T11:58:16.502119Z","shell.execute_reply.started":"2024-02-11T11:14:36.100830Z","shell.execute_reply":"2024-02-11T11:58:16.501246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo_format=dict(path=\"/kaggle/working/\",\n                 train=\"/kaggle/working/data/images/train\",\n                 val=\"/kaggle/working/data/images/val\",\n                 nc=1,\n                 train_labels='/kaggle/working/train_labels.csv',\n                 names={0:\"pneumonia\"})\n             \nwith open('/kaggle/working/yolo.yaml', 'w') as outfile:\n    yaml.dump(yolo_format, outfile, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:58:16.503608Z","iopub.execute_input":"2024-02-11T11:58:16.503975Z","iopub.status.idle":"2024-02-11T11:58:16.511586Z","shell.execute_reply.started":"2024-02-11T11:58:16.503936Z","shell.execute_reply":"2024-02-11T11:58:16.510584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2024-02-11T11:58:16.512841Z","iopub.execute_input":"2024-02-11T11:58:16.513121Z","iopub.status.idle":"2024-02-11T11:58:31.613201Z","shell.execute_reply.started":"2024-02-11T11:58:16.513098Z","shell.execute_reply":"2024-02-11T11:58:31.612151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO\n\nmodel = YOLO('yolov8m.pt')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T13:32:38.213460Z","iopub.execute_input":"2024-02-11T13:32:38.214318Z","iopub.status.idle":"2024-02-11T13:32:38.307345Z","shell.execute_reply.started":"2024-02-11T13:32:38.214274Z","shell.execute_reply":"2024-02-11T13:32:38.306286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freeze_layers = 20\nfor i, (name, param) in enumerate(model.named_parameters()):\n    if i < freeze_layers:\n        param.requires_grad = False\n\nmodel.train(data=\"/kaggle/working/yolo.yaml\",epochs=50,patience=5,batch=8,\n                    lr0=0.0005,imgsz=1024)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T14:41:44.336157Z","iopub.execute_input":"2024-02-11T14:41:44.336617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}