{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## RSNA Pneumonia Detection Challenge Data Cleaning\n<p>The RSNA Pneumonia Detection Challenge is a very old practice competition on Kaggle. If one wants to perform this with yolo, one would require to make certain changes to the data to prepare it as yaml for yolo. This notebook presents a way to change dicom to jpeg and csv to txt and further copying them to respective yaml structured folders</p>","metadata":{}},{"cell_type":"markdown","source":"### Install dependencies\nInstall them as a rule","metadata":{}},{"cell_type":"code","source":"%%capture\nimport os\n! pip install scikit-learn numpy matplotlib pandas lightning torch timm==0.5.4\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim import Adam\nimport lightning as L\nfrom torchvision import datasets, transforms\nimport pandas as pd\nimport numpy as np\nfrom glob import glob\nfrom PIL import Image\nimport logging\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-26T05:03:11.566621Z","iopub.execute_input":"2026-04-26T05:03:11.567164Z","iopub.status.idle":"2026-04-26T05:03:18.913904Z","shell.execute_reply.started":"2026-04-26T05:03:11.567090Z","shell.execute_reply":"2026-04-26T05:03:18.912583Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Label Change\nOne has to change the labels in the form of x,y,w,h. They are always normalized, so whichever value you give as size, it would not matter.","metadata":{}},{"cell_type":"code","source":"label_src=\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nout=\"label\"\nIMG_W = 1024\nIMG_H = 1024\nos.makedirs(out,exist_ok=True)\ndf=pd.read_csv(label_src)\nfor patientId,group in df.groupby(\"patientId\"):\n    label_path=os.path.join(out, f\"{patientId}.txt\")\n    with open(label_path,\"w\") as f:\n        for _, row in group.iterrows():\n            if row[\"Target\"] == 0:\n                continue\n\n            x = row[\"x\"]\n            y = row[\"y\"]\n            w = row[\"width\"]\n            h = row[\"height\"]\n            \n            x_center=(x + w/2)/IMG_W\n            y_center=(y + h/2)/IMG_H\n            w_norm=w/IMG_W\n            h_norm=h/IMG_H\n            f.write(f\"0 {x_center:.6f} {y_center:.6f} {w_norm:.6f} {h_norm:.6f}\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T05:03:18.916095Z","iopub.execute_input":"2026-04-26T05:03:18.916566Z","iopub.status.idle":"2026-04-26T05:03:29.184371Z","shell.execute_reply.started":"2026-04-26T05:03:18.916501Z","shell.execute_reply":"2026-04-26T05:03:29.183124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Image Change\nChange the images from DICOM to JPEG. Yolo needs images in jpeg","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport cv2\n\nDICOM_DIR=\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_images\"\nOUTPUT_DIR=\"/kaggle/working/rsna_yolo/images/train\"\n\nos.makedirs(OUTPUT_DIR,exist_ok=True)\n\nfor file in os.listdir(DICOM_DIR):\n    if not file.endswith(\".dcm\"):\n        continue\n\n    dcm_path = os.path.join(DICOM_DIR, file)\n\n    ds = pydicom.dcmread(dcm_path)\n    img = ds.pixel_array.astype(np.float32)\n\n    if hasattr(ds, \"PhotometricInterpretation\"):\n        if ds.PhotometricInterpretation == \"MONOCHROME1\":\n            img = np.max(img) - img\n    img = img - np.min(img)\n    if np.max(img) != 0:\n        img = img / np.max(img)\n    img = (img * 255).astype(np.uint8)\n    filename = file.replace(\".dcm\", \".png\")\n    save_path = os.path.join(OUTPUT_DIR, filename)\n    cv2.imwrite(save_path, img)\n\nprint(f\"Done. PNG images saved in: {OUTPUT_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T05:03:36.137945Z","iopub.execute_input":"2026-04-26T05:03:36.138477Z","iopub.status.idle":"2026-04-26T05:05:48.430571Z","shell.execute_reply.started":"2026-04-26T05:03:36.138439Z","shell.execute_reply":"2026-04-26T05:05:48.428769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### YOLO'S YAML FORMAT\nNext you  can reorder your files in this format:\n\n```yaml\nrsna_yolo/\n│\n├── images/\n│   ├── train/\n│   └── val/\n│\n├── labels/\n│   ├── train/\n│   └── val/\n│\n└── data.yaml\n```\n\nBecause YOLO needs it this way.\nUse this block of code for the purpose:\n\n```python\nimport shutil # Move files instead of copying\nimport random\n\nIMAGE_DIR = \"/kaggle/working/rsna_yolo/images\"\nLABEL_DIR = \"/kaggle/working/label\"\n\nBASE_DIR = \"/kaggle/working/rsna_yolo\"\n\nTRAIN_IMG_DIR = os.path.join(BASE_DIR, \"images/train\")\nVAL_IMG_DIR = os.path.join(BASE_DIR, \"images/val\")\n\nTRAIN_LABEL_DIR = os.path.join(BASE_DIR, \"labels/train\")\nVAL_LABEL_DIR = os.path.join(BASE_DIR, \"labels/val\")\n\nfor path in [\n    TRAIN_IMG_DIR,\n    VAL_IMG_DIR,\n    TRAIN_LABEL_DIR,\n    VAL_LABEL_DIR\n]:\n    os.makedirs(path, exist_ok=True)\n\nimages = [\n    f for f in os.listdir(IMAGE_DIR)\n    if f.endswith(\".png\")\n]\n\nrandom.seed(42)\nrandom.shuffle(images)\n\n# 70-30 split\nsplit_idx = int(0.7 * len(images))\n\ntrain_images = images[:split_idx]\nval_images = images[split_idx:]\n\ndef move_files(file_list, img_dest, label_dest):\n    for img_file in file_list:\n        base_name = os.path.splitext(img_file)[0]\n        label_file = base_name + \".txt\"\n\n        src_img = os.path.join(IMAGE_DIR, img_file)\n        src_label = os.path.join(LABEL_DIR, label_file)\n\n        dst_img = os.path.join(img_dest, img_file)\n        dst_label = os.path.join(label_dest, label_file)\n\n        if os.path.exists(src_img):\n            shutil.move(src_img, dst_img)\n\n        if os.path.exists(src_label):\n            shutil.move(src_label, dst_label)\n        else:\n            open(dst_label, \"w\").close()\n\nmove_files(train_images, TRAIN_IMG_DIR, TRAIN_LABEL_DIR)\nmove_files(val_images, VAL_IMG_DIR, VAL_LABEL_DIR)\n\nprint(\"Done.\")\nprint(f\"Train images: {len(train_images)}\")\nprint(f\"Val images: {len(val_images)}\")\n```","metadata":{}}]}