{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":6103,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":4646,"modelId":2804}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport os\nimport shutil\nimport cv2\nimport pydicom\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n# 경로 설정\nINPUT_PATH = Path(\"/kaggle/input/rsna-pneumonia-detection-challenge\")\nDCM_DIR = INPUT_PATH / \"stage_2_train_images\"\nTEST_DCM_DIR = INPUT_PATH / \"stage_2_test_images\"\nLABEL_CSV = INPUT_PATH / \"stage_2_train_labels.csv\"\nCLASS_CSV = INPUT_PATH / \"stage_2_detailed_class_info.csv\"\n\n# 출력 경로\nYOLO_IMG_DIR = Path(\"kaggle/working/yolo/images\")\nTRAIN_IMG_DIR = YOLO_IMG_DIR / \"train\"\nVAL_IMG_DIR = YOLO_IMG_DIR / \"val\"\nTEST_IMG_DIR = YOLO_IMG_DIR / \"test\"\nSAVE_PATH = Path(\"kaggle/working\")\n\nTRAIN_IMG_DIR.mkdir(parents=True, exist_ok=True)\nVAL_IMG_DIR.mkdir(parents=True, exist_ok=True)\nTEST_IMG_DIR.mkdir(parents=True, exist_ok=True)\n\nIMG_SIZE = 640\nJPEG_QUALITY = 95\n\nprint(\"complete import\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:50:56.142429Z","iopub.execute_input":"2025-06-04T22:50:56.142890Z","iopub.status.idle":"2025-06-04T22:50:56.148889Z","shell.execute_reply.started":"2025-06-04T22:50:56.142868Z","shell.execute_reply":"2025-06-04T22:50:56.148238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. 중복 제거 후 병합\nlabels_df = pd.read_csv(LABEL_CSV).drop_duplicates()\nclass_df = pd.read_csv(CLASS_CSV).drop_duplicates(subset=\"patientId\")\nmerged_df = labels_df.merge(class_df, on=\"patientId\", how=\"left\")\n\nprint(\"complete merge\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:43:38.058224Z","iopub.execute_input":"2025-06-04T22:43:38.058996Z","iopub.status.idle":"2025-06-04T22:43:38.217354Z","shell.execute_reply.started":"2025-06-04T22:43:38.058969Z","shell.execute_reply":"2025-06-04T22:43:38.216460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 메타데이터 추출 함수\ndef extract_dicom_meta(dcm_path):\n    try:\n        dcm = pydicom.dcmread(dcm_path, stop_before_pixels=True)\n        return {\n            \"PatientAge\": getattr(dcm, \"PatientAge\", \"\"),\n            \"BodyPartExamined\": getattr(dcm, \"BodyPartExamined\", \"\"),\n            \"ViewPosition\": getattr(dcm, \"ViewPosition\", \"\"),\n            \"PatientSex\": getattr(dcm, \"PatientSex\", \"\")\n        }\n    except:\n        return {\n            \"PatientAge\": \"\", \"BodyPartExamined\": \"\", \"ViewPosition\": \"\", \"PatientSex\": \"\"\n        }\n\n# 3. Train 메타데이터 추출\ntrain_meta_list = []\nfor pid in tqdm(merged_df[\"patientId\"].unique(), desc=\"Train 메타데이터 추출\"):\n    dcm_path = DCM_DIR / f\"{pid}.dcm\"\n    if not dcm_path.exists():\n        continue\n    meta = extract_dicom_meta(dcm_path)\n    meta[\"patientId\"] = pid\n    train_meta_list.append(meta)\n\ntrain_meta_df = pd.DataFrame(train_meta_list)\nmerged_df = merged_df.merge(train_meta_df, on=\"patientId\", how=\"left\")\n\nmerged_df.to_csv(SAVE_PATH / \"merged_all.csv\", index=False)\nprint(\"Train 병합 + 메타 저장 완료\")\n\n# 📌 3. Test 메타데이터 추출\ntest_meta_list = []\nfor pid in tqdm([f.stem for f in TEST_DCM_DIR.glob(\"*.dcm\")], desc=\"Test 메타데이터 추출\"):\n    dcm_path = TEST_DCM_DIR / f\"{pid}.dcm\"\n    if not dcm_path.exists():\n        continue\n    meta = extract_dicom_meta(dcm_path)\n    meta[\"patientId\"] = pid\n    test_meta_list.append(meta)\n\ntest_meta_df = pd.DataFrame(test_meta_list)\ntest_meta_df.to_csv(SAVE_PATH / \"test_metadata.csv\", index=False)\nprint(\"✅ Test 메타 저장 완료\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:51:00.480720Z","iopub.execute_input":"2025-06-04T22:51:00.481277Z","iopub.status.idle":"2025-06-04T22:52:24.421608Z","shell.execute_reply.started":"2025-06-04T22:51:00.481254Z","shell.execute_reply":"2025-06-04T22:52:24.420854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. train/val 분할\nmerged_df = pd.read_csv(SAVE_PATH / \"merged_all.csv\")\n\nunique_ids = merged_df[\"patientId\"].unique()\n\n# patientId 단위로 8:2 분할\ntrain_ids, val_ids = train_test_split(unique_ids, test_size=0.2, random_state=42)\n\ntrain_df = merged_df[merged_df[\"patientId\"].isin(train_ids)].copy()\nval_df = merged_df[merged_df[\"patientId\"].isin(val_ids)].copy()\n\ntrain_df.to_csv(SAVE_PATH / \"merged_train.csv\", index=False)\nval_df.to_csv(SAVE_PATH / \"merged_val.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:52:45.456097Z","iopub.execute_input":"2025-06-04T22:52:45.456842Z","iopub.status.idle":"2025-06-04T22:52:45.474034Z","shell.execute_reply.started":"2025-06-04T22:52:45.456815Z","shell.execute_reply":"2025-06-04T22:52:45.472851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. DICOM → JPG 변환 및 저장\ndef convert_dicom_to_jpg(patient_ids, dicom_dir, save_dir):\n    for pid in tqdm(patient_ids, desc=f\"JPG 변환 → {save_dir.name}\"):\n        dcm_path = dicom_dir / f\"{pid}.dcm\"\n        jpg_path = save_dir / f\"{pid}.jpg\"\n        if not dcm_path.exists():\n            continue\n        try:\n            dcm = pydicom.dcmread(dcm_path)\n            img = dcm.pixel_array.astype(\"float32\")\n            img = cv2.normalize(img, None, 0, 255, cv2.NORM_MINMAX).astype(\"uint8\")\n            resized = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n            cv2.imwrite(str(jpg_path), resized, [int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])\n        except Exception as e:\n            print(f\"⚠️ {pid} 변환 실패: {e}\")\n\n# 변환 실행\nconvert_dicom_to_jpg(train_df[\"patientId\"], DCM_DIR, TRAIN_IMG_DIR)\nconvert_dicom_to_jpg(val_df[\"patientId\"], DCM_DIR, VAL_IMG_DIR)\nconvert_dicom_to_jpg(test_df[\"patientId\"], TEST_DCM_DIR, TEST_IMG_DIR)\n\nprint(f\" 변환된 train 이미지 수: {len(list(TRAIN_IMG_DIR.glob('*.jpg')))}\")\nprint(f\" 변환된 val 이미지 수: {len(list(VAL_IMG_DIR.glob('*.jpg')))}\")\nprint(f\" 변환된 val 이미지 수: {len(list(TEST_IMG_DIR.glob('*.jpg')))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:52:52.058523Z","iopub.execute_input":"2025-06-04T22:52:52.058790Z","iopub.status.idle":"2025-06-04T22:52:54.944520Z","shell.execute_reply.started":"2025-06-04T22:52:52.058770Z","shell.execute_reply":"2025-06-04T22:52:54.943477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. 라벨 생성\ndef create_yolo_labels(df, label_dir, img_size=640):\n    label_dir.mkdir(parents=True, exist_ok=True)\n\n    for pid, group in tqdm(df.groupby(\"patientId\"), desc=f\"{label_dir.name} 라벨 생성\"):\n        lines = []\n        for _, row in group.iterrows():\n            if row[\"Target\"] == 1:\n                x = row[\"x\"]\n                y = row[\"y\"]\n                w = row[\"width\"]\n                h = row[\"height\"]\n\n                # 중심점 계산 + 정규화\n                x_center = (x + w / 2) / img_size\n                y_center = (y + h / 2) / img_size\n                w_norm = w / img_size\n                h_norm = h / img_size\n\n                # class_id는 pneumonia 1개 → 0으로 고정\n                lines.append(f\"0 {x_center:.6f} {y_center:.6f} {w_norm:.6f} {h_norm:.6f}\")\n\n        # 파일 저장\n        label_path = label_dir / f\"{pid}.txt\"\n        with open(label_path, \"w\") as f:\n            f.write(\"\\n\".join(lines))\n            \nTRAIN_LABEL_DIR = SAVE_PATH / \"yolo\" / \"labels\" / \"train\"\nVAL_LABEL_DIR = SAVE_PATH / \"yolo\" / \"labels\" / \"val\"\nTEST_LABEL_DIR = SAVE_PATH / \"yolo\" / \"labels\" / \"test\"\n\ncreate_yolo_labels(train_df, TRAIN_LABEL_DIR)\ncreate_yolo_labels(val_df, VAL_LABEL_DIR)\ncreate_yolo_labels(test_meta_df, TEST_LABEL_DIR)\n\nprint(\" YOLO 라벨 생성 완료\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:37:25.675564Z","iopub.execute_input":"2025-06-04T22:37:25.676145Z","iopub.status.idle":"2025-06-04T22:37:31.993658Z","shell.execute_reply.started":"2025-06-04T22:37:25.676119Z","shell.execute_reply":"2025-06-04T22:37:31.992892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. yaml 파일 생성\nimport yaml\n\nDATA_YAML_PATH = SAVE_PATH / \"yolo\" / \"data.yaml\"\n\ndata_yaml = {\n    \"path\": str((SAVE_PATH / \"yolo\").resolve()),\n    \"train\": \"images/train\",\n    \"val\": \"images/val\",\n    \"nc\": 1,\n    \"names\": [\"pneumonia\"]\n}\n\n# 저장\nwith open(DATA_YAML_PATH, \"w\") as f:\n    yaml.dump(data_yaml, f, default_flow_style=False)\n\nprint(f\"✅ data.yaml 생성 완료: {DATA_YAML_PATH}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T22:38:36.833416Z","iopub.execute_input":"2025-06-04T22:38:36.834158Z","iopub.status.idle":"2025-06-04T22:38:36.859203Z","shell.execute_reply.started":"2025-06-04T22:38:36.834132Z","shell.execute_reply":"2025-06-04T22:38:36.858567Z"}},"outputs":[],"execution_count":null}]}