{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":5,"nbformat":4,"cells":[{"id":"de7bc263-bf9f-487d-9ad4-2eafacbbc382","cell_type":"markdown","source":"# Extract only DICOM files referenced by `train_label_coordinates.csv`\n\nNotebook này sẽ:\n\n1. Đọc `train_label_coordinates.csv`.\n2. Lấy đúng các file `.dcm` được tham chiếu bởi `study_id`, `series_id`, `instance_number`.\n3. Loại bỏ file trùng lặp.\n4. Giữ cấu trúc thư mục `study_id/series_id/instance_number.dcm`.\n5. Xuất thêm file metadata và nén toàn bộ thành `.zip` để tải từ Kaggle Output.\n","metadata":{}},{"id":"4e444082-79d4-481c-9f7f-1bd6dbb0f43a","cell_type":"code","source":"from pathlib import Path\nimport shutil\nimport pandas as pd\n\n# Kaggle thường mount competition dataset tại /kaggle/input/<competition-slug>\nINPUT_BASE = Path('/kaggle/input')\nOUTPUT_DIR = Path('/kaggle/working/rsna_coordinate_dcm')\nZIP_PATH = Path('/kaggle/working/rsna_coordinate_dcm.zip')\n\n# Để None nếu muốn lấy toàn bộ bệnh nhân có trong train_label_coordinates.csv.\n# Ví dụ chỉ lấy một số bệnh nhân:\n# SELECTED_STUDY_IDS = [4003253, 4646740]\nSELECTED_STUDY_IDS = None\n\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nprint('Output directory:', OUTPUT_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:17:27.203099Z","iopub.execute_input":"2026-07-26T14:17:27.204048Z","iopub.status.idle":"2026-07-26T14:17:28.164968Z","shell.execute_reply.started":"2026-07-26T14:17:27.204009Z","shell.execute_reply":"2026-07-26T14:17:28.164163Z"}},"outputs":[{"name":"stdout","text":"Output directory: /kaggle/working/rsna_coordinate_dcm\n","output_type":"stream"}],"execution_count":1},{"id":"46653490-28ae-425c-834b-724391240b45","cell_type":"code","source":"# Tự tìm đúng thư mục dataset, tránh phụ thuộc cứng vào tên slug.\ncoordinate_csv_candidates = list(INPUT_BASE.rglob('train_label_coordinates.csv'))\n\nif not coordinate_csv_candidates:\n    raise FileNotFoundError(\n        'Không tìm thấy train_label_coordinates.csv trong /kaggle/input. '\n        'Hãy kiểm tra dataset đã được Add Input chưa.'\n    )\n\nfor idx, path in enumerate(coordinate_csv_candidates):\n    print(idx, path)\n\n# Với competition này thường chỉ có một file khớp.\nCOORDINATE_CSV = coordinate_csv_candidates[0]\nDATASET_ROOT = COORDINATE_CSV.parent\nTRAIN_IMAGES_DIR = DATASET_ROOT / 'train_images'\n\nif not TRAIN_IMAGES_DIR.exists():\n    raise FileNotFoundError(f'Không tìm thấy thư mục train_images tại: {TRAIN_IMAGES_DIR}')\n\nprint('\\nDataset root:', DATASET_ROOT)\nprint('Coordinates CSV:', COORDINATE_CSV)\nprint('Train images:', TRAIN_IMAGES_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:17:28.16596Z","iopub.execute_input":"2026-07-26T14:17:28.166291Z","iopub.status.idle":"2026-07-26T14:19:59.157438Z","shell.execute_reply.started":"2026-07-26T14:17:28.166267Z","shell.execute_reply":"2026-07-26T14:19:59.156824Z"}},"outputs":[{"name":"stdout","text":"0 /kaggle/input/competitions/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv\n\nDataset root: /kaggle/input/competitions/rsna-2024-lumbar-spine-degenerative-classification\nCoordinates CSV: /kaggle/input/competitions/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv\nTrain images: /kaggle/input/competitions/rsna-2024-lumbar-spine-degenerative-classification/train_images\n","output_type":"stream"}],"execution_count":2},{"id":"58ae2c23-46a7-4cad-a22b-bab5da67bdfd","cell_type":"code","source":"coords = pd.read_csv(COORDINATE_CSV)\n\nrequired_columns = {\n    'study_id', 'series_id', 'instance_number',\n    'condition', 'level', 'x', 'y'\n}\nmissing_columns = required_columns - set(coords.columns)\nif missing_columns:\n    raise ValueError(f'Thiếu các cột bắt buộc: {sorted(missing_columns)}')\n\nif SELECTED_STUDY_IDS is not None:\n    selected_ids = {int(x) for x in SELECTED_STUDY_IDS}\n    coords = coords[coords['study_id'].astype(int).isin(selected_ids)].copy()\n\nif coords.empty:\n    raise ValueError('Không còn dòng dữ liệu nào sau khi lọc study_id.')\n\nprint('Số dòng coordinate:', len(coords))\nprint('Số study:', coords['study_id'].nunique())\nprint('Số series:', coords[['study_id', 'series_id']].drop_duplicates().shape[0])\ndisplay(coords.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:19:59.1592Z","iopub.execute_input":"2026-07-26T14:19:59.159726Z","iopub.status.idle":"2026-07-26T14:19:59.38937Z","shell.execute_reply.started":"2026-07-26T14:19:59.159703Z","shell.execute_reply":"2026-07-26T14:19:59.388545Z"}},"outputs":[{"name":"stdout","text":"Số dòng coordinate: 48692\nSố study: 1974\nSố series: 6291\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"   study_id  series_id  instance_number              condition  level  \\\n0   4003253  702807833                8  Spinal Canal Stenosis  L1/L2   \n1   4003253  702807833                8  Spinal Canal Stenosis  L2/L3   \n2   4003253  702807833                8  Spinal Canal Stenosis  L3/L4   \n3   4003253  702807833                8  Spinal Canal Stenosis  L4/L5   \n4   4003253  702807833                8  Spinal Canal Stenosis  L5/S1   \n\n            x           y  \n0  322.831858  227.964602  \n1  320.571429  295.714286  \n2  323.030303  371.818182  \n3  335.292035  427.327434  \n4  353.415929  483.964602  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>study_id</th>\n      <th>series_id</th>\n      <th>instance_number</th>\n      <th>condition</th>\n      <th>level</th>\n      <th>x</th>\n      <th>y</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>Spinal Canal Stenosis</td>\n      <td>L1/L2</td>\n      <td>322.831858</td>\n      <td>227.964602</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>Spinal Canal Stenosis</td>\n      <td>L2/L3</td>\n      <td>320.571429</td>\n      <td>295.714286</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>Spinal Canal Stenosis</td>\n      <td>L3/L4</td>\n      <td>323.030303</td>\n      <td>371.818182</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>Spinal Canal Stenosis</td>\n      <td>L4/L5</td>\n      <td>335.292035</td>\n      <td>427.327434</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>Spinal Canal Stenosis</td>\n      <td>L5/S1</td>\n      <td>353.415929</td>\n      <td>483.964602</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":3},{"id":"a1c958e5-7de3-4d26-9fd1-a24218ec3d4e","cell_type":"code","source":"# Một file DICOM có thể được nhiều dòng coordinate tham chiếu,\n# ví dụ cùng một sagittal slice chứa nhiều level.\n# Vì vậy chỉ copy mỗi file duy nhất một lần.\nunique_dicom_refs = (\n    coords[['study_id', 'series_id', 'instance_number']]\n    .drop_duplicates()\n    .reset_index(drop=True)\n)\n\ncopy_records = []\nmissing_files = []\n\nfor row in unique_dicom_refs.itertuples(index=False):\n    study_id = str(int(row.study_id))\n    series_id = str(int(row.series_id))\n    instance_number = str(int(row.instance_number))\n\n    src = TRAIN_IMAGES_DIR / study_id / series_id / f'{instance_number}.dcm'\n    dst = OUTPUT_DIR / 'train_images' / study_id / series_id / f'{instance_number}.dcm'\n\n    if not src.exists():\n        missing_files.append({\n            'study_id': study_id,\n            'series_id': series_id,\n            'instance_number': instance_number,\n            'expected_path': str(src),\n        })\n        continue\n\n    dst.parent.mkdir(parents=True, exist_ok=True)\n    shutil.copy2(src, dst)\n\n    copy_records.append({\n        'study_id': int(study_id),\n        'series_id': int(series_id),\n        'instance_number': int(instance_number),\n        'source_path': str(src),\n        'output_path': str(dst),\n    })\n\nprint(f'Đã copy: {len(copy_records)} file DICOM duy nhất')\nprint(f'Thiếu: {len(missing_files)} file')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:19:59.390265Z","iopub.execute_input":"2026-07-26T14:19:59.390549Z","iopub.status.idle":"2026-07-26T14:23:49.682149Z","shell.execute_reply.started":"2026-07-26T14:19:59.390517Z","shell.execute_reply":"2026-07-26T14:23:49.681346Z"}},"outputs":[{"name":"stdout","text":"Đã copy: 24546 file DICOM duy nhất\nThiếu: 0 file\n","output_type":"stream"}],"execution_count":4},{"id":"d0cd967c-5de3-4768-803a-14ea92e6c666","cell_type":"code","source":"# Lưu coordinate annotation gốc tương ứng với các bệnh nhân đã chọn.\ncoords_output_path = OUTPUT_DIR / 'train_label_coordinates_filtered.csv'\ncoords.to_csv(coords_output_path, index=False)\n\n# Lưu danh sách file đã copy.\nmanifest = pd.DataFrame(copy_records)\nmanifest_output_path = OUTPUT_DIR / 'dicom_manifest.csv'\nmanifest.to_csv(manifest_output_path, index=False)\n\n# Nếu có file không tìm thấy thì lưu log riêng.\nif missing_files:\n    pd.DataFrame(missing_files).to_csv(\n        OUTPUT_DIR / 'missing_dicom_files.csv',\n        index=False\n    )\n\nprint('Saved:', coords_output_path)\nprint('Saved:', manifest_output_path)\ndisplay(manifest.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:23:49.683946Z","iopub.execute_input":"2026-07-26T14:23:49.684247Z","iopub.status.idle":"2026-07-26T14:23:50.171607Z","shell.execute_reply.started":"2026-07-26T14:23:49.684223Z","shell.execute_reply":"2026-07-26T14:23:50.171046Z"}},"outputs":[{"name":"stdout","text":"Saved: /kaggle/working/rsna_coordinate_dcm/train_label_coordinates_filtered.csv\nSaved: /kaggle/working/rsna_coordinate_dcm/dicom_manifest.csv\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"   study_id   series_id  instance_number  \\\n0   4003253   702807833                8   \n1   4003253  1054713880                4   \n2   4003253  1054713880                5   \n3   4003253  1054713880                6   \n4   4003253  1054713880               11   \n\n                                         source_path  \\\n0  /kaggle/input/competitions/rsna-2024-lumbar-sp...   \n1  /kaggle/input/competitions/rsna-2024-lumbar-sp...   \n2  /kaggle/input/competitions/rsna-2024-lumbar-sp...   \n3  /kaggle/input/competitions/rsna-2024-lumbar-sp...   \n4  /kaggle/input/competitions/rsna-2024-lumbar-sp...   \n\n                                         output_path  \n0  /kaggle/working/rsna_coordinate_dcm/train_imag...  \n1  /kaggle/working/rsna_coordinate_dcm/train_imag...  \n2  /kaggle/working/rsna_coordinate_dcm/train_imag...  \n3  /kaggle/working/rsna_coordinate_dcm/train_imag...  \n4  /kaggle/working/rsna_coordinate_dcm/train_imag...  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>study_id</th>\n      <th>series_id</th>\n      <th>instance_number</th>\n      <th>source_path</th>\n      <th>output_path</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>4003253</td>\n      <td>702807833</td>\n      <td>8</td>\n      <td>/kaggle/input/competitions/rsna-2024-lumbar-sp...</td>\n      <td>/kaggle/working/rsna_coordinate_dcm/train_imag...</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>4003253</td>\n      <td>1054713880</td>\n      <td>4</td>\n      <td>/kaggle/input/competitions/rsna-2024-lumbar-sp...</td>\n      <td>/kaggle/working/rsna_coordinate_dcm/train_imag...</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>4003253</td>\n      <td>1054713880</td>\n      <td>5</td>\n      <td>/kaggle/input/competitions/rsna-2024-lumbar-sp...</td>\n      <td>/kaggle/working/rsna_coordinate_dcm/train_imag...</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>4003253</td>\n      <td>1054713880</td>\n      <td>6</td>\n      <td>/kaggle/input/competitions/rsna-2024-lumbar-sp...</td>\n      <td>/kaggle/working/rsna_coordinate_dcm/train_imag...</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>4003253</td>\n      <td>1054713880</td>\n      <td>11</td>\n      <td>/kaggle/input/competitions/rsna-2024-lumbar-sp...</td>\n      <td>/kaggle/working/rsna_coordinate_dcm/train_imag...</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":5},{"id":"896e62fe-3692-4584-9474-e1b77a96fdf8","cell_type":"code","source":"# Nén thư mục để tải về thuận tiện từ phần Output của Kaggle Notebook.\nif ZIP_PATH.exists():\n    ZIP_PATH.unlink()\n\narchive_path = shutil.make_archive(\n    base_name=str(ZIP_PATH.with_suffix('')),\n    format='zip',\n    root_dir=OUTPUT_DIR\n)\n\nzip_size_mb = Path(archive_path).stat().st_size / (1024 ** 2)\nprint(f'ZIP created: {archive_path}')\nprint(f'ZIP size: {zip_size_mb:.2f} MB')\nprint('\\nSau khi chạy xong: Save Version → mở tab Output → tải file rsna_coordinate_dcm.zip')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:23:50.172866Z","iopub.execute_input":"2026-07-26T14:23:50.173207Z","iopub.status.idle":"2026-07-26T14:28:34.714221Z","shell.execute_reply.started":"2026-07-26T14:23:50.173182Z","shell.execute_reply":"2026-07-26T14:28:34.713351Z"}},"outputs":[{"name":"stdout","text":"ZIP created: /kaggle/working/rsna_coordinate_dcm.zip\nZIP size: 4790.88 MB\n\nSau khi chạy xong: Save Version → mở tab Output → tải file rsna_coordinate_dcm.zip\n","output_type":"stream"}],"execution_count":6},{"id":"4ea6dfd9-1a12-4843-ba6d-ebf902f43df9","cell_type":"markdown","source":"## Cấu trúc file đầu ra\n\n```text\nrsna_coordinate_dcm/\n├── train_images/\n│   └── study_id/\n│       └── series_id/\n│           └── instance_number.dcm\n├── train_label_coordinates_filtered.csv\n├── dicom_manifest.csv\n└── missing_dicom_files.csv   # chỉ xuất hiện nếu có file thiếu\n```\n\n`rsna_coordinate_dcm.zip` sẽ nằm trong `/kaggle/working/`.\n","metadata":{}},{"id":"5f1ff245-f433-4d7f-8dfa-4d60fed9f6f8","cell_type":"code","source":"from pathlib import Path\nfrom IPython.display import HTML, display\n\nzip_path = Path(\"/kaggle/working/rsna_coordinate_dcm.zip\")\n\nif not zip_path.exists():\n    raise FileNotFoundError(f\"Không tìm thấy file: {zip_path}\")\n\nsize_mb = zip_path.stat().st_size / (1024 ** 2)\n\ndisplay(\n    HTML(\n        f\"\"\"\n        <p>File đã sẵn sàng, dung lượng: <b>{size_mb:.2f} MB</b></p>\n        <a href=\"files/{zip_path.name}\" download>\n            Tải rsna_coordinate_dcm.zip\n        </a>\n        \"\"\"\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T14:34:54.349544Z","iopub.execute_input":"2026-07-26T14:34:54.350192Z","iopub.status.idle":"2026-07-26T14:34:54.356692Z","shell.execute_reply.started":"2026-07-26T14:34:54.350162Z","shell.execute_reply":"2026-07-26T14:34:54.356003Z"}},"outputs":[{"output_type":"display_data","data":{"text/plain":"<IPython.core.display.HTML object>","text/html":"\n        <p>File đã sẵn sàng, dung lượng: <b>4790.88 MB</b></p>\n        <a href=\"files/rsna_coordinate_dcm.zip\" download>\n            Tải rsna_coordinate_dcm.zip\n        </a>\n        "},"metadata":{}}],"execution_count":8}]}