{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":10943904,"sourceType":"datasetVersion","datasetId":6806501}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:13.584027Z","iopub.execute_input":"2025-04-25T03:53:13.584330Z","iopub.status.idle":"2025-04-25T03:53:13.589036Z","shell.execute_reply.started":"2025-04-25T03:53:13.584309Z","shell.execute_reply":"2025-04-25T03:53:13.588057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DIR = \"/kaggle/input/rsna-lumbar-metadata/data/processed_metadata\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:13.796586Z","iopub.execute_input":"2025-04-25T03:53:13.796853Z","iopub.status.idle":"2025-04-25T03:53:13.801095Z","shell.execute_reply.started":"2025-04-25T03:53:13.796834Z","shell.execute_reply":"2025-04-25T03:53:13.800106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:14.016856Z","iopub.execute_input":"2025-04-25T03:53:14.017754Z","iopub.status.idle":"2025-04-25T03:53:14.335332Z","shell.execute_reply.started":"2025-04-25T03:53:14.017717Z","shell.execute_reply":"2025-04-25T03:53:14.334605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"left = pd.read_csv(f\"{DIR}/processed_metadata_LeftNeuralForaminalNarrowing.csv\")\nright = pd.read_csv(f\"{DIR}/processed_metadata_RightNeuralForaminalNarrowing.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:14.336436Z","iopub.execute_input":"2025-04-25T03:53:14.336786Z","iopub.status.idle":"2025-04-25T03:53:14.464004Z","shell.execute_reply.started":"2025-04-25T03:53:14.336758Z","shell.execute_reply":"2025-04-25T03:53:14.463304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"left.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:14.465234Z","iopub.execute_input":"2025-04-25T03:53:14.465547Z","iopub.status.idle":"2025-04-25T03:53:14.471779Z","shell.execute_reply.started":"2025-04-25T03:53:14.465518Z","shell.execute_reply":"2025-04-25T03:53:14.471105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"right.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:14.656015Z","iopub.execute_input":"2025-04-25T03:53:14.656298Z","iopub.status.idle":"2025-04-25T03:53:14.661638Z","shell.execute_reply.started":"2025-04-25T03:53:14.656278Z","shell.execute_reply":"2025-04-25T03:53:14.660879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"left.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:14.870812Z","iopub.execute_input":"2025-04-25T03:53:14.871156Z","iopub.status.idle":"2025-04-25T03:53:14.898757Z","shell.execute_reply.started":"2025-04-25T03:53:14.871131Z","shell.execute_reply":"2025-04-25T03:53:14.897817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"paths_df1 = set(left['image_path'].unique())\npaths_df2 = set(right['image_path'].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:15.131512Z","iopub.execute_input":"2025-04-25T03:53:15.131771Z","iopub.status.idle":"2025-04-25T03:53:15.145856Z","shell.execute_reply.started":"2025-04-25T03:53:15.131752Z","shell.execute_reply":"2025-04-25T03:53:15.144850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_paths = paths_df1.intersection(paths_df2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:15.325795Z","iopub.execute_input":"2025-04-25T03:53:15.326075Z","iopub.status.idle":"2025-04-25T03:53:15.330713Z","shell.execute_reply.started":"2025-04-25T03:53:15.326054Z","shell.execute_reply":"2025-04-25T03:53:15.329992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Total unique paths in df1: {len(paths_df1)}\")\nprint(f\"Total unique paths in df2: {len(paths_df2)}\")\nprint(f\"Number of duplicate image paths across both: {len(duplicate_paths)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:15.515491Z","iopub.execute_input":"2025-04-25T03:53:15.515772Z","iopub.status.idle":"2025-04-25T03:53:15.520809Z","shell.execute_reply.started":"2025-04-25T03:53:15.515752Z","shell.execute_reply":"2025-04-25T03:53:15.519932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if duplicate_paths:\n    print(\"Sample duplicate paths:\")\n    for path in list(duplicate_paths)[:5]:  # Show just a few for sanity check\n        print(path)\nelse:\n    print(\"No duplicate image paths found between the two DataFrames.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:15.732380Z","iopub.execute_input":"2025-04-25T03:53:15.732701Z","iopub.status.idle":"2025-04-25T03:53:15.737682Z","shell.execute_reply.started":"2025-04-25T03:53:15.732680Z","shell.execute_reply":"2025-04-25T03:53:15.736850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"left_sub = pd.read_csv(f\"{DIR}/processed_metadata_LeftSubarticularStenosis.csv\")\nright_sub = pd.read_csv(f\"{DIR}/processed_metadata_RightSubarticularStenosis.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:15.968033Z","iopub.execute_input":"2025-04-25T03:53:15.968883Z","iopub.status.idle":"2025-04-25T03:53:16.070295Z","shell.execute_reply.started":"2025-04-25T03:53:15.968857Z","shell.execute_reply":"2025-04-25T03:53:16.069511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"paths_df1 = set(left['image_path'].unique())\npaths_df2 = set(right['image_path'].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:16.215837Z","iopub.execute_input":"2025-04-25T03:53:16.216114Z","iopub.status.idle":"2025-04-25T03:53:16.225959Z","shell.execute_reply.started":"2025-04-25T03:53:16.216096Z","shell.execute_reply":"2025-04-25T03:53:16.225115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_paths = paths_df1.intersection(paths_df2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:16.397990Z","iopub.execute_input":"2025-04-25T03:53:16.398507Z","iopub.status.idle":"2025-04-25T03:53:16.402549Z","shell.execute_reply.started":"2025-04-25T03:53:16.398483Z","shell.execute_reply":"2025-04-25T03:53:16.401616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Total unique paths in df1: {len(paths_df1)}\")\nprint(f\"Total unique paths in df2: {len(paths_df2)}\")\nprint(f\"Number of duplicate image paths across both: {len(duplicate_paths)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:16.624103Z","iopub.execute_input":"2025-04-25T03:53:16.624759Z","iopub.status.idle":"2025-04-25T03:53:16.629398Z","shell.execute_reply.started":"2025-04-25T03:53:16.624736Z","shell.execute_reply":"2025-04-25T03:53:16.628559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"left = pd.read_csv(f\"{DIR}/processed_metadata_LeftNeuralForaminalNarrowing.csv\")\nright = pd.read_csv(f\"{DIR}/processed_metadata_RightNeuralForaminalNarrowing.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:16.816332Z","iopub.execute_input":"2025-04-25T03:53:16.817009Z","iopub.status.idle":"2025-04-25T03:53:16.892933Z","shell.execute_reply.started":"2025-04-25T03:53:16.816982Z","shell.execute_reply":"2025-04-25T03:53:16.892053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"severity_counts_df1 = left['severity_code'].value_counts().sort_index()\nseverity_counts_df2 = right['severity_code'].value_counts().sort_index()\nprint(severity_counts_df1)\nprint(severity_counts_df2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:20.216687Z","iopub.execute_input":"2025-04-25T03:53:20.216963Z","iopub.status.idle":"2025-04-25T03:53:20.229643Z","shell.execute_reply.started":"2025-04-25T03:53:20.216944Z","shell.execute_reply":"2025-04-25T03:53:20.228772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"severity_counts_df1_sub = left_sub['severity_code'].value_counts().sort_index()\nseverity_counts_df2_sub = right_sub['severity_code'].value_counts().sort_index()\nprint(severity_counts_df1_sub)\nprint(severity_counts_df2_sub)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:20.614728Z","iopub.execute_input":"2025-04-25T03:53:20.615002Z","iopub.status.idle":"2025-04-25T03:53:20.622513Z","shell.execute_reply.started":"2025-04-25T03:53:20.614981Z","shell.execute_reply":"2025-04-25T03:53:20.621531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"spinal = pd.read_csv(f\"{DIR}/processed_metadata_SpinalCanalStenosis.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:20.998715Z","iopub.execute_input":"2025-04-25T03:53:20.998987Z","iopub.status.idle":"2025-04-25T03:53:21.058277Z","shell.execute_reply.started":"2025-04-25T03:53:20.998968Z","shell.execute_reply":"2025-04-25T03:53:21.057514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"severity_counts_df_spinal = spinal['severity_code'].value_counts().sort_index()\nprint(severity_counts_df_spinal)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T03:53:21.385186Z","iopub.execute_input":"2025-04-25T03:53:21.385455Z","iopub.status.idle":"2025-04-25T03:53:21.391932Z","shell.execute_reply.started":"2025-04-25T03:53:21.385436Z","shell.execute_reply":"2025-04-25T03:53:21.391111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Augmentation","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport albumentations as A\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:08:54.403629Z","iopub.execute_input":"2025-04-25T05:08:54.403976Z","iopub.status.idle":"2025-04-25T05:08:54.409121Z","shell.execute_reply.started":"2025-04-25T05:08:54.403950Z","shell.execute_reply":"2025-04-25T05:08:54.408151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CONFIG ---\nTARGET_TOTAL = 15000\nCLASS_COUNTS = {0: 5000, 1: 5000, 2: 5000}\nROTATION_ANGLES = [0.5, -0.5, 1.0, -1.0, 1.5, -1.5]\n\n# Define input/output paths\nDIR = \"/kaggle/input/rsna-lumbar-metadata/data/processed_metadata\"\nINPUT_CSV = f\"{DIR}/processed_metadata_LeftNeuralForaminalNarrowing.csv\"\nOUTPUT_FOLDER = \"/kaggle/working/augmented/left_neural\"\nos.makedirs(OUTPUT_FOLDER, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:08:55.120000Z","iopub.execute_input":"2025-04-25T05:08:55.120340Z","iopub.status.idle":"2025-04-25T05:08:55.126140Z","shell.execute_reply.started":"2025-04-25T05:08:55.120317Z","shell.execute_reply":"2025-04-25T05:08:55.125168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Load CSV ---\ndf = pd.read_csv(INPUT_CSV)\n\n# --- Load DICOM and normalize ---\ndef load_image(path):\n    try:\n        dicom = pydicom.dcmread(path)\n        image = dicom.pixel_array.astype(np.float32)\n        image -= np.min(image)\n        image /= np.max(image)\n        image = (image * 255).astype(np.uint8)\n        return image\n    except Exception as e:\n        print(f\"[ERROR] Failed to load DICOM: {path} | {e}\")\n        return None\n\n# --- Save DICOM with augmented pixel data ---\ndef save_dicom_image(original_path, image_array, save_path):\n    try:\n        # Read the original DICOM\n        ds = pydicom.dcmread(original_path)\n\n        # Extract relevant DICOM information\n        study_id = ds.StudyInstanceUID\n        series_id = ds.SeriesInstanceUID\n        instance_no = ds.InstanceNumber\n\n        # Ensure correct type and dimensions\n        if image_array.dtype != np.uint16:\n            image_array = image_array.astype(np.uint16)\n\n        # Update the DICOM metadata\n        ds.SamplesPerPixel = 1\n        ds.PhotometricInterpretation = \"MONOCHROME2\"  # for grayscale images\n        ds.Rows, ds.Columns = image_array.shape\n        ds.BitsAllocated = 16\n        ds.BitsStored = 16\n        ds.HighBit = 15  # Highest bit of a 16-bit image\n        ds.PixelRepresentation = 0  # Unsigned integers\n        ds.file_meta.TransferSyntaxUID = pydicom.uid.ExplicitVRLittleEndian\n\n        # Set the PixelData to the image (ensure it's in bytes)\n        ds.PixelData = image_array.tobytes()\n\n        # Update SOP Instance UID and MediaStorage SOP Instance UID to prevent conflicts\n        ds.SOPInstanceUID = pydicom.uid.generate_uid()\n        ds.file_meta.MediaStorageSOPInstanceUID = ds.SOPInstanceUID\n\n        # --- Modify the filename to include study_id, series_id, and instance_no ---\n        base_name = os.path.basename(save_path)  # Get the base name of the save path (e.g., file name)\n        new_filename = f\"aug_{study_id}_series{series_id}_inst{instance_no}_aug{base_name}\"\n        new_path = os.path.join(os.path.dirname(save_path), new_filename)\n\n        # Save the modified DICOM\n        ds.save_as(new_path)\n        # print(f\"✅ Saved DICOM with 16-bit depth at {new_path}\")\n    except Exception as e:\n        print(f\"[ERROR] Saving DICOM failed for {save_path} | {e}\")\n\n\n\n# --- Rotation Transform ---\ndef get_rotation_transform(angle):\n    return A.Compose([\n        A.Rotate(limit=(angle, angle), border_mode=cv2.BORDER_REFLECT, p=1.0)\n    ])\n\n# --- Augment & Save as DICOM ---\ndef augment_and_save(images, required_count, class_id, base_df):\n    saved_rows = []\n    augmented = 0\n    i = 0\n    while augmented < required_count:\n        image_path = images[i % len(images)]\n        image = load_image(image_path)\n        if image is None:\n            i += 1\n            continue\n\n        for angle in ROTATION_ANGLES:\n            if augmented >= required_count:\n                break\n            transform = get_rotation_transform(angle)\n            augmented_image = transform(image=image)['image']\n\n            new_filename = f\"aug_sev{class_id}_{augmented}.dcm\"\n            new_path = os.path.join(OUTPUT_FOLDER, new_filename)\n            save_dicom_image(image_path, augmented_image, new_path)\n\n            row = base_df[base_df['image_path'] == image_path].iloc[0].copy()\n            row['image_path'] = new_path\n            saved_rows.append(row)\n            augmented += 1\n        i += 1\n    return saved_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:09:39.341817Z","iopub.execute_input":"2025-04-25T05:09:39.342156Z","iopub.status.idle":"2025-04-25T05:09:39.389108Z","shell.execute_reply.started":"2025-04-25T05:09:39.342132Z","shell.execute_reply":"2025-04-25T05:09:39.388308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Create balanced dataset ---\nbalanced_df = []\n\n# 🟢 Handle Normal (0)\nnormal_df = df[df['severity_code'] == 0].sample(CLASS_COUNTS[0], random_state=42)\nfor idx, row in tqdm(normal_df.iterrows(), total=len(normal_df), desc=\"Processing Normal\"):\n    image = load_image(row['image_path'])\n    if image is None:\n        continue\n    new_filename = f\"normal_{idx}.dcm\"\n    new_path = os.path.join(OUTPUT_FOLDER, new_filename)\n    save_dicom_image(row['image_path'], image, new_path)\n    row['image_path'] = new_path\n    balanced_df.append(row)\n\n# 🟡 Handle Moderate (1)\nmoderate_df = df[df['severity_code'] == 1]\nmoderate_images = moderate_df['image_path'].tolist()\nneeded = CLASS_COUNTS[1] - len(moderate_images)\nbalanced_df += moderate_df.to_dict('records')\nbalanced_df += augment_and_save(moderate_images, needed, 1, moderate_df)\n\n# 🔴 Handle Severe (2)\nsevere_df = df[df['severity_code'] == 2]\nsevere_images = severe_df['image_path'].tolist()\nneeded = CLASS_COUNTS[2] - len(severe_images)\nbalanced_df += severe_df.to_dict('records')\nbalanced_df += augment_and_save(severe_images, needed, 2, severe_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:10:16.800551Z","iopub.execute_input":"2025-04-25T05:10:16.800900Z","iopub.status.idle":"2025-04-25T05:13:03.256430Z","shell.execute_reply.started":"2025-04-25T05:10:16.800877Z","shell.execute_reply":"2025-04-25T05:13:03.255445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_csv = '/kaggle/working/balanced_dataset.csv'\ncleaned_balanced_df = [pd.Series(item) for item in balanced_df]\ndf = pd.DataFrame(cleaned_balanced_df)\ndf.to_csv(output_csv, index=False)\nprint(f\"✅ Balanced dataset saved to: {output_csv}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:13:18.490381Z","iopub.execute_input":"2025-04-25T05:13:18.491377Z","iopub.status.idle":"2025-04-25T05:13:20.143954Z","shell.execute_reply.started":"2025-04-25T05:13:18.491346Z","shell.execute_reply":"2025-04-25T05:13:20.142885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"balanced_df[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:13:23.243118Z","iopub.execute_input":"2025-04-25T05:13:23.243428Z","iopub.status.idle":"2025-04-25T05:13:23.250068Z","shell.execute_reply.started":"2025-04-25T05:13:23.243405Z","shell.execute_reply":"2025-04-25T05:13:23.249093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:13:25.184485Z","iopub.execute_input":"2025-04-25T05:13:25.184976Z","iopub.status.idle":"2025-04-25T05:13:25.205716Z","shell.execute_reply.started":"2025-04-25T05:13:25.184934Z","shell.execute_reply":"2025-04-25T05:13:25.204639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:13:32.087663Z","iopub.execute_input":"2025-04-25T05:13:32.088477Z","iopub.status.idle":"2025-04-25T05:13:32.093833Z","shell.execute_reply.started":"2025-04-25T05:13:32.088435Z","shell.execute_reply":"2025-04-25T05:13:32.092704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"severity_counts_df1 = df['severity_code'].value_counts().sort_index()\nprint(severity_counts_df1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:13:45.783145Z","iopub.execute_input":"2025-04-25T05:13:45.783530Z","iopub.status.idle":"2025-04-25T05:13:45.790002Z","shell.execute_reply.started":"2025-04-25T05:13:45.783501Z","shell.execute_reply":"2025-04-25T05:13:45.789173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport pydicom\n\n# --- CONFIG ---\nDATA_DIR = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\nAUG_DIR = \"/kaggle/working/augmented/left_neural\"\nimage_dir = os.path.join(DATA_DIR, 'train_images')\n\n# 🔧 Function to generate original DICOM image paths\ndef generate_image_paths(study_id, series_id):\n    series_dir = os.path.join(image_dir, str(study_id), str(series_id))\n    images = sorted(os.listdir(series_dir))\n    image_paths = [os.path.join(series_dir, img) for img in images]\n    return image_paths\n\n# 🔍 Function to construct the augmented filename using DICOM metadata\ndef get_augmented_path_from_dicom(orig_path):\n    try:\n        ds = pydicom.dcmread(orig_path)\n        study_id = ds.StudyInstanceUID\n        series_id = ds.SeriesInstanceUID\n        instance_no = ds.InstanceNumber\n        base_name = os.path.basename(orig_path)\n        aug_filename = f\"aug_{study_id}_series{series_id}_inst{instance_no}_aug{base_name}\"\n        aug_path = os.path.join(AUG_DIR, aug_filename)\n        return aug_path if os.path.exists(aug_path) else None\n    except Exception as e:\n        print(f\"[ERROR] Could not read DICOM: {orig_path} | {e}\")\n        return None\n\n# 🖼️ Display original vs augmented images using new naming convention\ndef display_original_vs_augmented(image_paths):\n    n = min(len(image_paths), 6)  # Limit to first few for quick view\n    plt.figure(figsize=(10, 5 * n))\n\n    for i in range(n):\n        orig_path = image_paths[i]\n        aug_path = get_augmented_path_from_dicom(orig_path)\n\n        if not aug_path:\n            print(f\"[WARN] No augmented file found for: {os.path.basename(orig_path)}\")\n            continue\n\n        try:\n            orig_ds = pydicom.dcmread(orig_path)\n            aug_ds = pydicom.dcmread(aug_path)\n\n            # Original\n            plt.subplot(n, 2, 2 * i + 1)\n            plt.imshow(orig_ds.pixel_array, cmap='gray')\n            plt.title(f\"Original: {os.path.basename(orig_path)}\")\n            plt.axis('off')\n\n            # Augmented\n            plt.subplot(n, 2, 2 * i + 2)\n            plt.imshow(aug_ds.pixel_array, cmap='gray')\n            plt.title(f\"Augmented: {os.path.basename(aug_path)}\")\n            plt.axis('off')\n        except Exception as e:\n            print(f\"[ERROR] Failed to load image pair {i}: {e}\")\n            continue\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:44:58.973412Z","iopub.execute_input":"2025-04-25T05:44:58.973831Z","iopub.status.idle":"2025-04-25T05:44:58.992780Z","shell.execute_reply.started":"2025-04-25T05:44:58.973797Z","shell.execute_reply":"2025-04-25T05:44:58.991717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_id = \"3617698707\"\nseries_id = \"3406806779\"\nimage_paths = generate_image_paths(study_id, series_id)\ndisplay_original_vs_augmented(image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:45:05.134975Z","iopub.execute_input":"2025-04-25T05:45:05.135778Z","iopub.status.idle":"2025-04-25T05:45:05.163728Z","shell.execute_reply.started":"2025-04-25T05:45:05.135750Z","shell.execute_reply":"2025-04-25T05:45:05.162784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"balanced_df[0]['image_path']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T05:45:11.753626Z","iopub.execute_input":"2025-04-25T05:45:11.754301Z","iopub.status.idle":"2025-04-25T05:45:11.760094Z","shell.execute_reply.started":"2025-04-25T05:45:11.754276Z","shell.execute_reply":"2025-04-25T05:45:11.759275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}