{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path where Kaggle mounts the competition dataset\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n# List what’s inside\nprint(os.listdir(data_path))\n\n# Load train.csv\ntrain_df = pd.read_csv(os.path.join(data_path, \"train.csv\"))\n\ntrain_df.shape\ntrain_df.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:11.881744Z","iopub.execute_input":"2025-10-13T09:47:11.882022Z","iopub.status.idle":"2025-10-13T09:47:11.959515Z","shell.execute_reply.started":"2025-10-13T09:47:11.882002Z","shell.execute_reply":"2025-10-13T09:47:11.958720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_coordinates = pd.read_csv(os.path.join(data_path, \"train_label_coordinates.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:12.866291Z","iopub.execute_input":"2025-10-13T09:47:12.866608Z","iopub.status.idle":"2025-10-13T09:47:13.097197Z","shell.execute_reply.started":"2025-10-13T09:47:12.866584Z","shell.execute_reply":"2025-10-13T09:47:13.096453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_coordinates","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:22.110105Z","iopub.execute_input":"2025-10-13T09:47:22.110371Z","iopub.status.idle":"2025-10-13T09:47:22.123771Z","shell.execute_reply.started":"2025-10-13T09:47:22.110349Z","shell.execute_reply":"2025-10-13T09:47:22.123114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n# list a few study folders\nprint(os.listdir(os.path.join(data_path, \"train_images\"))[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:28.364897Z","iopub.execute_input":"2025-10-13T09:47:28.365695Z","iopub.status.idle":"2025-10-13T09:47:28.408607Z","shell.execute_reply.started":"2025-10-13T09:47:28.365661Z","shell.execute_reply":"2025-10-13T09:47:28.407934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_id = \"4003253\"\nstudy_path = os.path.join(data_path, \"train_images\", study_id)\n\nprint(\"Series inside study:\", os.listdir(study_path)[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:28.409564Z","iopub.execute_input":"2025-10-13T09:47:28.409804Z","iopub.status.idle":"2025-10-13T09:47:28.423080Z","shell.execute_reply.started":"2025-10-13T09:47:28.409786Z","shell.execute_reply":"2025-10-13T09:47:28.422580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_id = os.listdir(study_path)[0]  # take first series\nseries_path = os.path.join(study_path, series_id)\n\nprint(\"DICOM slices inside series:\", os.listdir(series_path)[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:36.593138Z","iopub.execute_input":"2025-10-13T09:47:36.593467Z","iopub.status.idle":"2025-10-13T09:47:36.611655Z","shell.execute_reply.started":"2025-10-13T09:47:36.593428Z","shell.execute_reply":"2025-10-13T09:47:36.611079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\ndcm_file = os.path.join(series_path, os.listdir(series_path)[0])  # first slice\ndcm = pydicom.dcmread(dcm_file)\n\nprint(dcm)  # metadata\nplt.imshow(dcm.pixel_array, cmap='gray')\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T06:00:35.260601Z","iopub.execute_input":"2025-10-13T06:00:35.260793Z","iopub.status.idle":"2025-10-13T06:00:35.723249Z","shell.execute_reply.started":"2025-10-13T06:00:35.260778Z","shell.execute_reply":"2025-10-13T06:00:35.722594Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nimport os\n\n# Pick one study and series (from your earlier code)\nstudy_id = \"4003253\"\nseries_id = os.listdir(os.path.join(data_path, \"train_images\", study_id))[0]\nseries_path = os.path.join(data_path, \"train_images\", study_id, series_id)\n\n# Get all slices in this series and sort by filename (important!)\ndcm_files = sorted(os.listdir(series_path))\n\n# Pick 5 slices spread across the series\nsample_slices = [dcm_files[i] for i in [0, len(dcm_files)//4, len(dcm_files)//2, 3*len(dcm_files)//4, -1]]\n\n# Plot them\nfig, axes = plt.subplots(1, 5, figsize=(20, 5))\n\nfor ax, fname in zip(axes, sample_slices):\n    dcm_path = os.path.join(series_path, fname)\n    dcm = pydicom.dcmread(dcm_path)\n    ax.imshow(dcm.pixel_array, cmap='gray')\n    ax.set_title(fname)\n    ax.axis('off')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T06:00:35.724616Z","iopub.execute_input":"2025-10-13T06:00:35.725084Z","iopub.status.idle":"2025-10-13T06:00:36.196049Z","shell.execute_reply.started":"2025-10-13T06:00:35.725059Z","shell.execute_reply":"2025-10-13T06:00:36.195404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\ncoords_path = os.path.join(data_path, \"train_label_coordinates.csv\")\ncoords_df = pd.read_csv(coords_path)\n\nprint(coords_df.shape)\nprint(coords_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:38:09.681867Z","iopub.execute_input":"2025-10-13T05:38:09.682367Z","iopub.status.idle":"2025-10-13T05:38:09.748624Z","shell.execute_reply.started":"2025-10-13T05:38:09.682337Z","shell.execute_reply":"2025-10-13T05:38:09.748079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nimport os\n\n# Pick one row from coords_df\nrow = coords_df.iloc[0]\n\nstudy_id = str(row.study_id)\nseries_id = str(row.series_id)\ninstance_number = int(row.instance_number)\n\n# Build path to the series folder\nseries_path = os.path.join(data_path, \"train_images\", study_id, series_id)\n\n# Get all slices (files) in this series\ndcm_files = sorted(os.listdir(series_path))\n\n# instance_number in metadata starts from 1, so adjust index\ndcm_file = os.path.join(series_path, dcm_files[instance_number - 1])\n\n# Load the DICOM slice\ndcm = pydicom.dcmread(dcm_file)\nimg = dcm.pixel_array\n\n# Plot image with coordinate marked\nplt.figure(figsize=(6,6))\nplt.imshow(img, cmap='gray')\nplt.scatter(row.x, row.y, c='red', s=40, label=f\"{row.condition} {row.level}\")\nplt.legend()\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:38:09.806673Z","iopub.execute_input":"2025-10-13T05:38:09.807071Z","iopub.status.idle":"2025-10-13T05:38:09.996808Z","shell.execute_reply.started":"2025-10-13T05:38:09.807055Z","shell.execute_reply":"2025-10-13T05:38:09.996061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\n# Pick a study to visualize\nstudy_id = \"4003253\"\n\n# Filter coords for this study\nstudy_coords = coords_df[coords_df.study_id == int(study_id)]\n\n# Pick one series from this study\nseries_id = str(study_coords.series_id.iloc[0])\nseries_path = os.path.join(data_path, \"train_images\", study_id, series_id)\n\n# Get all slices in this series\ndcm_files = sorted(os.listdir(series_path))\n\n# Choose a slice number that has multiple labels (for demonstration)\nslice_num = study_coords.instance_number.iloc[0]\ndcm_file = os.path.join(series_path, dcm_files[slice_num - 1])\n\n# Load the slice\ndcm = pydicom.dcmread(dcm_file)\nimg = dcm.pixel_array\n\n# Plot slice with all coordinates from this slice\nplt.figure(figsize=(7,7))\nplt.imshow(img, cmap='gray')\n\nfor _, row in study_coords[study_coords.instance_number == slice_num].iterrows():\n    plt.scatter(row.x, row.y, c='red', s=40)\n    plt.text(row.x+5, row.y+5, f\"{row.level}\", color='yellow', fontsize=9)\n\nplt.title(f\"Study {study_id} - Slice {slice_num}\")\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T03:21:48.705695Z","iopub.execute_input":"2025-10-13T03:21:48.706014Z","iopub.status.idle":"2025-10-13T03:21:48.957360Z","shell.execute_reply.started":"2025-10-13T03:21:48.705995Z","shell.execute_reply":"2025-10-13T03:21:48.956568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_study_ids = set(train_df.study_id.astype(str))\nimage_study_ids = set(os.listdir(os.path.join(data_path, \"train_images\")))\n\nmissing_studies = train_study_ids - image_study_ids\nextra_studies = image_study_ids - train_study_ids\n\nprint(\"Missing studies in images:\", len(missing_studies))\nprint(\"Extra studies not in train.csv:\", len(extra_studies))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:49.935050Z","iopub.execute_input":"2025-10-13T09:47:49.935347Z","iopub.status.idle":"2025-10-13T09:47:49.943869Z","shell.execute_reply.started":"2025-10-13T09:47:49.935319Z","shell.execute_reply":"2025-10-13T09:47:49.943185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Base dataset path\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\ndef count_dicom_files(base_dir):\n    total_studies = 0\n    total_series = 0\n    total_dicom = 0\n    \n    for study in os.listdir(base_dir):\n        study_path = os.path.join(base_dir, study)\n        if not os.path.isdir(study_path):\n            continue\n        total_studies += 1\n        for series in os.listdir(study_path):\n            series_path = os.path.join(study_path, series)\n            if not os.path.isdir(series_path):\n                continue\n            total_series += 1\n            total_dicom += len([f for f in os.listdir(series_path) if f.endswith(\".dcm\")])\n    \n    return total_studies, total_series, total_dicom\n\n\n# Train data\ntrain_dir = os.path.join(data_path, \"train_images\")\ntrain_stats = count_dicom_files(train_dir)\nprint(f\"🧠 TRAIN SET → Studies: {train_stats[0]} | Series: {train_stats[1]} | DICOMs: {train_stats[2]}\")\n\n# Test data\ntest_dir = os.path.join(data_path, \"test_images\")\ntest_stats = count_dicom_files(test_dir)\nprint(f\"🧪 TEST SET  → Studies: {test_stats[0]} | Series: {test_stats[1]} | DICOMs: {test_stats[2]}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:47:50.381631Z","iopub.execute_input":"2025-10-13T09:47:50.381864Z","iopub.status.idle":"2025-10-13T09:49:50.652714Z","shell.execute_reply.started":"2025-10-13T09:47:50.381847Z","shell.execute_reply":"2025-10-13T09:49:50.652048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport pandas as pd\n\n# Path to dataset\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\ntrain_img_dir = os.path.join(data_path, \"train_images\")\n\n# List all study folders\nall_studies = sorted([s for s in os.listdir(train_img_dir) if os.path.isdir(os.path.join(train_img_dir, s))])\n\n# Randomly select 2 studies for test\nrandom.seed(42)\ntest_studies = random.sample(all_studies, 2)\ntrain_studies = [s for s in all_studies if s not in test_studies]\n\n# Save splits\npd.DataFrame({\"study_id\": train_studies}).to_csv(\"/kaggle/working/train_split_studies.csv\", index=False)\npd.DataFrame({\"study_id\": test_studies}).to_csv(\"/kaggle/working/test_split_studies.csv\", index=False)\n\nprint(\"✅ Logical split created.\")\nprint(f\"Train studies: {len(train_studies)}\")\nprint(f\"Test studies : {len(test_studies)}\")\nprint(\"📄 Files saved: train_split_studies.csv & test_split_studies.csv\")\nprint(\"📁 Test studies:\", test_studies)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:52:29.357940Z","iopub.execute_input":"2025-10-13T09:52:29.358589Z","iopub.status.idle":"2025-10-13T09:52:30.565852Z","shell.execute_reply.started":"2025-10-13T09:52:29.358559Z","shell.execute_reply":"2025-10-13T09:52:30.565210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"coords = pd.read_csv(os.path.join(data_path, \"train_label_coordinates.csv\"))\ntrain_ids = pd.read_csv(\"/kaggle/working/train_split_studies.csv\")[\"study_id\"].astype(str)\ntest_ids = pd.read_csv(\"/kaggle/working/test_split_studies.csv\")[\"study_id\"].astype(str)\n\ntrain_labels = coords[coords[\"study_id\"].astype(str).isin(train_ids)]\ntest_labels  = coords[coords[\"study_id\"].astype(str).isin(test_ids)]\n\nprint(train_labels.shape, test_labels.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:52:34.210837Z","iopub.execute_input":"2025-10-13T09:52:34.211107Z","iopub.status.idle":"2025-10-13T09:52:34.313008Z","shell.execute_reply.started":"2025-10-13T09:52:34.211086Z","shell.execute_reply":"2025-10-13T09:52:34.312219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nfrom glob import glob\nfrom tqdm import tqdm\n\nBASE_PATH = data_path\nLABEL_CSV = os.path.join(BASE_PATH, \"train_label_coordinates.csv\")\nTRAIN_DIR = os.path.join(BASE_PATH, \"train_images\")\n\n# Load labels\ndf = pd.read_csv(LABEL_CSV)\n\n# Use study_id + series_id from folder structure\nseries_index = {}\n\nprint(\"Indexing DICOM files...\")\ndicom_files = glob(os.path.join(TRAIN_DIR, \"**\", \"*.dcm\"), recursive=True)\n\nfor fp in tqdm(dicom_files):\n    try:\n        ds = pydicom.dcmread(fp, stop_before_pixels=True)\n        inst_no = int(ds.InstanceNumber)\n\n        # Extract study_id and series_id from folder path\n        parts = fp.split(os.sep)\n        study_id = parts[-3]\n        series_id = parts[-2]\n        key = (study_id, series_id)\n\n        if key not in series_index:\n            series_index[key] = set()\n        series_index[key].add(inst_no)\n    except Exception:\n        continue\n\nprint(f\"Indexed {len(series_index)} (study_id, series_id) pairs from {len(dicom_files)} DICOMs.\")\n\n# Now check labels against index\nmissing_series = []\nmissing_instances = []\n\nprint(\"Checking label references...\")\nfor idx, row in tqdm(df.iterrows(), total=len(df)):\n    study_id = str(row[\"study_id\"])\n    series_id = str(row[\"series_id\"])\n    inst_no = int(row[\"instance_number\"])\n    key = (study_id, series_id)\n\n    if key not in series_index:\n        missing_series.append((idx, study_id, series_id))\n    elif inst_no not in series_index[key]:\n        missing_instances.append((idx, study_id, series_id, inst_no))\n\nprint(f\"Total rows checked: {len(df)}\")\nprint(f\"Missing series: {len(missing_series)}\")\nprint(f\"Missing instances: {len(missing_instances)}\")\n\n# Save reports\npd.DataFrame(missing_series, columns=[\"row_index\",\"study_id\",\"series_id\"]).to_csv(\"/kaggle/working/missing_series.csv\", index=False)\npd.DataFrame(missing_instances, columns=[\"row_index\",\"study_id\",\"series_id\",\"instance_number\"]).to_csv(\"/kaggle/working/missing_instances.csv\", index=False)\n\nprint(\"Reports saved in /kaggle/working/: missing_series.csv, missing_instances.csv\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normalise and Cropping","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\nimport cv2\nfrom tqdm import tqdm\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Base path (same as before)\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n# Parameters\nPATCH_SIZE = 128\nOUTPUT_DIR = \"/kaggle/working/patches_windowed\"\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# Load label coordinates\ncoords_df = pd.read_csv(os.path.join(data_path, \"train_label_coordinates.csv\"))\n\n# ============== Helper Functions ==============\n\ndef window_image(img, center, width):\n    \"\"\"Apply DICOM windowing to convert 16-bit -> 8-bit visible range.\"\"\"\n    img_min = center - width / 2\n    img_max = center + width / 2\n    img = np.clip(img, img_min, img_max)\n    img = (img - img_min) / (img_max - img_min)\n    img = (img * 255).astype(np.uint8)\n    return img\n\ndef safe_crop(img, center_x, center_y, size=PATCH_SIZE):\n    \"\"\"Crop square patch centered at (x,y) safely even near borders.\"\"\"\n    h, w = img.shape\n    half = size // 2\n    x1, x2 = center_x - half, center_x + half\n    y1, y2 = center_y - half, center_y + half\n    patch = np.zeros((size, size), dtype=img.dtype)\n    x1_img, x2_img = max(0, x1), min(w, x2)\n    y1_img, y2_img = max(0, y1), min(h, y2)\n    x1_patch, y1_patch = x1_img - x1, y1_img - y1\n    x2_patch, y2_patch = x1_patch + (x2_img - x1_img), y1_patch + (y2_img - y1_img)\n    patch[y1_patch:y2_patch, x1_patch:x2_patch] = img[y1_img:y2_img, x1_img:x2_img]\n    return patch\n\n# ============== Cropping Loop ==============\n\nsaved, skipped = 0, 0\nsample_patches = []  # to visualize\n\nfor idx, row in tqdm(coords_df.iterrows(), total=len(coords_df)):\n    study_id = str(row[\"study_id\"])\n    series_id = str(row[\"series_id\"])\n    inst_no = int(row[\"instance_number\"])\n    x, y = int(row[\"x\"]), int(row[\"y\"])\n\n    condition = str(row['condition']).replace(\" \", \"_\")\n    level = str(row['level']).replace(\"/\", \"-\")\n\n    series_path = os.path.join(data_path, \"train_images\", study_id, series_id)\n\n    try:\n        dcm_file = os.path.join(series_path, f\"{inst_no}.dcm\")\n        dcm = pydicom.dcmread(dcm_file)\n        img = dcm.pixel_array.astype(np.float32)\n\n        # Apply DICOM windowing\n        wc = float(getattr(dcm, \"WindowCenter\", 300))\n        ww = float(getattr(dcm, \"WindowWidth\", 600))\n        img = window_image(img, wc, ww)\n\n        patch = safe_crop(img, x, y, PATCH_SIZE)\n\n        # Skip if patch is nearly blank (to avoid pure black)\n        if np.mean(patch) < 5:\n            skipped += 1\n            continue\n\n        out_path = os.path.join(OUTPUT_DIR, f\"{study_id}_{series_id}_{inst_no}_{condition}_{level}.png\")\n        ok = cv2.imwrite(out_path, patch)\n\n        if ok:\n            saved += 1\n            # collect a few random samples for preview\n            if len(sample_patches) < 5 and np.random.rand() < 0.001:\n                sample_patches.append((patch, f\"{condition}_{level}\"))\n        else:\n            skipped += 1\n    except Exception as e:\n        skipped += 1\n        continue\n\nprint(f\"✅ Done. Saved: {saved} patches | Skipped: {skipped}\")\nprint(f\"✅ Output dir: {OUTPUT_DIR}\")\n\n# ============== Quick Visualization ==============\n\nif sample_patches:\n    plt.figure(figsize=(15,3))\n    for i, (patch, title) in enumerate(sample_patches):\n        plt.subplot(1, len(sample_patches), i+1)\n        plt.imshow(patch, cmap='gray')\n        plt.title(title)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"⚠️ No sample patches collected for preview. (Try increasing probability)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T09:52:50.475881Z","iopub.execute_input":"2025-10-13T09:52:50.476145Z","iopub.status.idle":"2025-10-13T10:03:36.100251Z","shell.execute_reply.started":"2025-10-13T09:52:50.476123Z","shell.execute_reply":"2025-10-13T10:03:36.099536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Paths\nPATCH_DIR = \"/kaggle/working/patches_windowed\"\ntrain_studies_path = \"/kaggle/working/train_split_studies.csv\"\ntest_studies_path = \"/kaggle/working/test_split_studies.csv\"\n\n# Read splits\ntrain_studies = pd.read_csv(train_studies_path)[\"study_id\"].astype(str).tolist()\ntest_studies  = pd.read_csv(test_studies_path)[\"study_id\"].astype(str).tolist()\n\n# Get all patch filenames\npatch_files = os.listdir(PATCH_DIR)\n\n# Build dataframe of patch metadata from filenames\ndata = []\nfor f in patch_files:\n    # filename example: 4003253_702807833_8_Spinal_Canal_Stenosis_L1-L2.png\n    parts = f.replace(\".png\", \"\").split(\"_\")\n    study_id, series_id, inst_no = parts[0], parts[1], parts[2]\n    condition = parts[3]\n    level = parts[4]\n    data.append([f, study_id, series_id, inst_no, condition, level])\n\npatch_df = pd.DataFrame(data, columns=[\"filename\", \"study_id\", \"series_id\", \"instance_number\", \"condition\", \"level\"])\n\n# Split based on study_id\ntrain_patches = patch_df[patch_df[\"study_id\"].isin(train_studies)]\ntest_patches  = patch_df[patch_df[\"study_id\"].isin(test_studies)]\n\n# Save split info\ntrain_patches.to_csv(\"/kaggle/working/train_patches.csv\", index=False)\ntest_patches.to_csv(\"/kaggle/working/test_patches.csv\", index=False)\n\nprint(f\"✅ Logical split complete!\")\nprint(f\"Train patches: {len(train_patches)} | Test patches: {len(test_patches)}\")\nprint(f\"📄 Saved: train_patches.csv and test_patches.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T10:15:56.646740Z","iopub.execute_input":"2025-10-13T10:15:56.647265Z","iopub.status.idle":"2025-10-13T10:15:56.988365Z","shell.execute_reply.started":"2025-10-13T10:15:56.647218Z","shell.execute_reply":"2025-10-13T10:15:56.987525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nPATCH_DIR = \"/kaggle/working/patches_windowed\"\n\n# List all patches\nall_patches = os.listdir(PATCH_DIR)\nprint(f\"Total patches found: {len(all_patches)}\")\n\n# Randomly sample up to 12 patches for visualization\nsample_files = random.sample(all_patches, min(12, len(all_patches)))\n\nplt.figure(figsize=(12, 8))\nfor i, fname in enumerate(sample_files):\n    img_path = os.path.join(PATCH_DIR, fname)\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n    plt.subplot(3, 4, i + 1)\n    plt.imshow(img, cmap='gray')\n    plt.title(fname.split(\"_\")[3])  # condition\n    plt.axis('off')\nplt.tight_layout()\nplt.show()\n\n# ---------- Black Pixel Check ----------\ndef is_black_patch(img, threshold=10, black_ratio=0.9):\n    \"\"\"\n    Returns True if more than 90% of pixels are below intensity 10.\n    \"\"\"\n    return np.mean(img < threshold) > black_ratio\n\nblack_count = 0\ncheck_limit = min(1000, len(all_patches))  # check up to 1000 randomly\n\nfor fname in random.sample(all_patches, check_limit):\n    img_path = os.path.join(PATCH_DIR, fname)\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n    if img is not None and is_black_patch(img):\n        black_count += 1\n\nblack_ratio = black_count / check_limit * 100\nprint(f\"\\n🧩 Quality check summary:\")\nprint(f\"Checked {check_limit} random patches\")\nprint(f\"Black/empty patches: {black_count} ({black_ratio:.2f}%)\")\n\nif black_ratio > 20:\n    print(\"⚠️ Warning: Too many black patches! You may need to adjust crop or window settings.\")\nelse:\n    print(\"✅ Looks good! Most patches contain visible anatomical content.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# train data preparation","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# =========================\n# Paths\n# =========================\ndata_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\ntrain_split_path = \"/kaggle/working/train_split_studies.csv\"\ntest_split_path  = \"/kaggle/working/test_split_studies.csv\"\n\n# =========================\n# Load Core CSVs\n# =========================\ncoords_df = pd.read_csv(os.path.join(data_path, \"train_label_coordinates.csv\"))\ntrain_df = pd.read_csv(os.path.join(data_path, \"train.csv\"))\n\n# Ensure all study_ids are strings for consistent merging\ncoords_df[\"study_id\"] = coords_df[\"study_id\"].astype(str)\ntrain_df[\"study_id\"] = train_df[\"study_id\"].astype(str)\n\n# =========================\n# Filter only your TRAIN split\n# =========================\ntrain_studies = pd.read_csv(train_split_path)[\"study_id\"].astype(str).tolist()\ncoords_df = coords_df[coords_df[\"study_id\"].isin(train_studies)]\n\nprint(f\"Using {len(coords_df)} labeled coordinates from {len(train_studies)} studies in train split\")\n\n# =========================\n# Define function: map condition + level → column in train.csv\n# =========================\ndef get_column_name(row):\n    cond = row[\"condition\"]\n    lvl = row[\"level\"].replace(\"/\", \"_\")\n    \n    if \"Spinal Canal\" in cond:\n        return f\"spinal_canal_stenosis_{lvl.lower()}\"\n    elif \"Left Neural\" in cond:\n        return f\"left_neural_foraminal_narrowing_{lvl.lower()}\"\n    elif \"Right Neural\" in cond:\n        return f\"right_neural_foraminal_narrowing_{lvl.lower()}\"\n    elif \"Left Subarticular\" in cond:\n        return f\"left_subarticular_stenosis_{lvl.lower()}\"\n    elif \"Right Subarticular\" in cond:\n        return f\"right_subarticular_stenosis_{lvl.lower()}\"\n    else:\n        return None\n\ncoords_df[\"target_col\"] = coords_df.apply(get_column_name, axis=1)\n\n# =========================\n# Merge severity information\n# =========================\nmerged_df = coords_df.merge(\n    train_df,\n    on=\"study_id\",\n    how=\"left\"\n)\n\n# Extract actual severity per label\nmerged_df[\"severity\"] = merged_df.apply(\n    lambda r: r[r[\"target_col\"]] if pd.notnull(r[\"target_col\"]) and r[\"target_col\"] in r.index else None,\n    axis=1\n)\n\nmerged_df = merged_df[[\"study_id\", \"series_id\", \"instance_number\", \"condition\", \"level\", \"x\", \"y\", \"severity\"]]\n\nprint(\"✅ Merging complete!\")\nprint(\"Unique severity levels:\", merged_df[\"severity\"].unique())\nprint(\"Example rows:\")\ndisplay(merged_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T10:17:22.747513Z","iopub.execute_input":"2025-10-13T10:17:22.747762Z","iopub.status.idle":"2025-10-13T10:17:23.763825Z","shell.execute_reply.started":"2025-10-13T10:17:22.747746Z","shell.execute_reply":"2025-10-13T10:17:23.763197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# =========================\n# Paths\n# =========================\nPATCH_DIR = \"/kaggle/working/patches_windowed\"\n\n\n# =========================\n# Get all patch filenames\n# =========================\npatch_files = os.listdir(PATCH_DIR)\npatch_df = pd.DataFrame({\"filename\": patch_files})\n\n# =========================\n# Extract IDs from filenames\n# Format: studyid_seriesid_instno_condition_level.png\n# =========================\npatch_df[\"study_id\"] = patch_df[\"filename\"].apply(lambda x: x.split(\"_\")[0])\npatch_df[\"series_id\"] = patch_df[\"filename\"].apply(lambda x: x.split(\"_\")[1])\npatch_df[\"instance_number\"] = patch_df[\"filename\"].apply(lambda x: int(x.split(\"_\")[2]))\n\n# 🛠 Ensure matching dtypes\nmerged_df[\"study_id\"] = merged_df[\"study_id\"].astype(str)\nmerged_df[\"series_id\"] = merged_df[\"series_id\"].astype(str)\npatch_df[\"study_id\"] = patch_df[\"study_id\"].astype(str)\npatch_df[\"series_id\"] = patch_df[\"series_id\"].astype(str)\n\n# =========================\n# Merge patches with labels\n# =========================\ntrain_ready = patch_df.merge(\n    merged_df[[\"study_id\", \"series_id\", \"instance_number\", \"severity\"]],\n    on=[\"study_id\", \"series_id\", \"instance_number\"],\n    how=\"left\"\n)\n\n# Drop unlabeled rows (if any)\ntrain_ready = train_ready.dropna(subset=[\"severity\"]).reset_index(drop=True)\n\nprint(\"✅ Patches matched with labels:\", len(train_ready))\nprint(\"🔸 Unique severity labels:\", train_ready[\"severity\"].unique())\ntrain_ready.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T10:17:32.676310Z","iopub.execute_input":"2025-10-13T10:17:32.676968Z","iopub.status.idle":"2025-10-13T10:17:32.915532Z","shell.execute_reply.started":"2025-10-13T10:17:32.676943Z","shell.execute_reply":"2025-10-13T10:17:32.914748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode string labels → numeric\nseverity_map = {\n    \"Normal/Mild\": 0,\n    \"Moderate\": 1,\n    \"Severe\": 2\n}\ntrain_ready[\"severity_encoded\"] = train_ready[\"severity\"].map(severity_map)\n\nprint(train_ready[\"severity_encoded\"].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T11:14:46.555740Z","iopub.execute_input":"2025-10-13T11:14:46.556433Z","iopub.status.idle":"2025-10-13T11:14:46.570317Z","shell.execute_reply.started":"2025-10-13T11:14:46.556407Z","shell.execute_reply":"2025-10-13T11:14:46.569559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\n\n# Define custom dataset\nclass SpinePatchDataset(Dataset):\n    def __init__(self, dataframe, image_dir, transform=None):\n        self.dataframe = dataframe\n        self.image_dir = image_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        row = self.dataframe.iloc[idx]\n        img_path = os.path.join(self.image_dir, row[\"filename\"])\n        image = Image.open(img_path).convert(\"L\")  # Convert to grayscale\n        label = row[\"severity_encoded\"]\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n# Define image transformations\ntransform = transforms.Compose([\n    transforms.Resize((128, 128)),           # Ensure consistent size\n    transforms.ToTensor(),                   # Convert to tensor\n    transforms.Normalize([0.5], [0.5])       # Normalize grayscale to [-1, 1]\n])\n\n# Split into train and validation\nfrom sklearn.model_selection import train_test_split\ntrain_df, val_df = train_test_split(train_ready, test_size=0.15, stratify=train_ready[\"severity_encoded\"], random_state=42)\n\n# Dataset instances\ntrain_dataset = SpinePatchDataset(train_df, PATCH_DIR, transform=transform)\nval_dataset = SpinePatchDataset(val_df, PATCH_DIR, transform=transform)\n\n# Dataloaders\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=2)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ResNet152","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom PIL import Image\nfrom tqdm import tqdm\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# ==============================\n# 1. CONFIG\n# ==============================\nPATCH_DIR = \"/kaggle/working/patches_windowed\"\nBATCH_SIZE = 32\nEPOCHS = 10\nLR = 1e-4\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# ==============================\n# 2. TRAIN / VAL SPLIT\n# ==============================\ntrain_df, val_df = train_test_split(\n    train_ready,\n    test_size=0.2,\n    stratify=train_ready[\"severity_encoded\"],\n    random_state=42\n)\n\n# ==============================\n# 3. DATASET CLASS\n# ==============================\nclass PatchDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(PATCH_DIR, row[\"filename\"])\n        img = Image.open(img_path).convert(\"RGB\")\n        label = int(row[\"severity_encoded\"])\n        \n        if self.transform:\n            img = self.transform(img)\n        return img, label\n\n\n# ==============================\n# 4. TRANSFORMS\n# ==============================\ntrain_tfms = transforms.Compose([\n    transforms.Resize((224,224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485,0.456,0.406], std=[0.229,0.224,0.225]),\n])\n\nval_tfms = transforms.Compose([\n    transforms.Resize((224,224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485,0.456,0.406], std=[0.229,0.224,0.225]),\n])\n\n# Datasets\ntrain_ds = PatchDataset(train_df, transform=train_tfms)\nval_ds = PatchDataset(val_df, transform=val_tfms)\n\n# ==============================\n# 5. CLASS WEIGHTS\n# ==============================\nclasses = np.unique(train_ready[\"severity_encoded\"])\nweights = compute_class_weight(class_weight=\"balanced\", classes=classes, y=train_ready[\"severity_encoded\"])\nclass_weights = torch.tensor(weights, dtype=torch.float).to(device)\nprint(\"Class Weights:\", class_weights)\n\n# ==============================\n# 6. DATALOADERS\n# ==============================\ntrain_dl = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True, num_workers=2)\nval_dl = DataLoader(val_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\n\n# ==============================\n# 7. MODEL: RESNET152\n# ==============================\nmodel = models.resnet152(pretrained=True)\nmodel.fc = nn.Linear(model.fc.in_features, 3)  # 3 classes: 0,1,2\nmodel = model.to(device)\n\ncriterion = nn.CrossEntropyLoss(weight=class_weights)\noptimizer = optim.Adam(model.parameters(), lr=LR)\n\n# ==============================\n# 8. TRAINING LOOP\n# ==============================\ntrain_loss_list, val_loss_list = [], []\ntrain_acc_list, val_acc_list = [], []\n\nfor epoch in range(EPOCHS):\n    model.train()\n    train_loss, correct, total = 0, 0, 0\n\n    for imgs, labels in tqdm(train_dl, desc=f\"Epoch {epoch+1}/{EPOCHS}\"):\n        imgs, labels = imgs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        \n        outputs = model(imgs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item() * imgs.size(0)\n        _, preds = torch.max(outputs, 1)\n        correct += (preds == labels).sum().item()\n        total += labels.size(0)\n\n    epoch_train_loss = train_loss / total\n    epoch_train_acc = correct / total\n\n    # Validation\n    model.eval()\n    val_loss, correct, total = 0, 0, 0\n    with torch.no_grad():\n        for imgs, labels in val_dl:\n            imgs, labels = imgs.to(device), labels.to(device)\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            val_loss += loss.item() * imgs.size(0)\n            _, preds = torch.max(outputs, 1)\n            correct += (preds == labels).sum().item()\n            total += labels.size(0)\n\n    epoch_val_loss = val_loss / total\n    epoch_val_acc = correct / total\n\n    train_loss_list.append(epoch_train_loss)\n    val_loss_list.append(epoch_val_loss)\n    train_acc_list.append(epoch_train_acc)\n    val_acc_list.append(epoch_val_acc)\n\n    print(f\"\\nEpoch [{epoch+1}/{EPOCHS}]\")\n    print(f\"Train Loss: {epoch_train_loss:.4f} | Train Acc: {epoch_train_acc:.4f}\")\n    print(f\"Val Loss: {epoch_val_loss:.4f} | Val Acc: {epoch_val_acc:.4f}\")\n\n# ==============================\n# 9. SAVE MODEL\n# ==============================\ntorch.save(model.state_dict(), \"/kaggle/working/resnet152_weighted.pth\")\nprint(\"✅ Model saved as resnet152_weighted.pth\")\n\n# ==============================\n# 10. PLOT METRICS\n# ==============================\nplt.figure(figsize=(12,5))\nplt.subplot(1,2,1)\nplt.plot(train_loss_list, label=\"Train Loss\")\nplt.plot(val_loss_list, label=\"Val Loss\")\nplt.legend(); plt.title(\"Loss per Epoch\")\n\nplt.subplot(1,2,2)\nplt.plot(train_acc_list, label=\"Train Acc\")\nplt.plot(val_acc_list, label=\"Val Acc\")\nplt.legend(); plt.title(\"Accuracy per Epoch\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T11:16:35.248203Z","iopub.execute_input":"2025-10-13T11:16:35.248507Z","iopub.status.idle":"2025-10-13T11:20:16.648544Z","shell.execute_reply.started":"2025-10-13T11:16:35.248485Z","shell.execute_reply":"2025-10-13T11:20:16.647171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}