{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Imports and Data Loading\nimport os\nimport time\nimport glob\nimport json\nimport collections\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as snsa\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport torchvision.models as models\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.patches as patches\nfrom matplotlib import animation, rc\n\n# For CV and metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (confusion_matrix, precision_score, recall_score,\n                             roc_curve, auc, accuracy_score, roc_auc_score)\nfrom sklearn.preprocessing import label_binarize\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Define the base path and read CSV files\ntrain_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\n\ntrain  = pd.read_csv(os.path.join(train_path, 'train.csv'))\nlabel = pd.read_csv(os.path.join(train_path, 'train_label_coordinates.csv'))\ntrain_desc  = pd.read_csv(os.path.join(train_path, 'train_series_descriptions.csv'))\ntest_desc   = pd.read_csv(os.path.join(train_path, 'test_series_descriptions.csv'))\nsub         = pd.read_csv(os.path.join(train_path, 'sample_submission.csv'))\n\n# Quick check of the dataframes\nprint(\"Train.csv head:\")\nprint(train.head(5))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:34:23.667976Z","iopub.execute_input":"2025-05-25T05:34:23.668344Z","iopub.status.idle":"2025-05-25T05:34:37.517642Z","shell.execute_reply.started":"2025-05-25T05:34:23.668316Z","shell.execute_reply":"2025-05-25T05:34:37.516460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef generate_image_paths(df, data_dir):\n    image_paths = []\n    for study_id, series_id in zip(df['study_id'], df['series_id']):\n        study_dir = os.path.join(data_dir, str(study_id))\n        series_dir = os.path.join(study_dir, str(series_id))\n        if os.path.exists(series_dir):\n            images = os.listdir(series_dir)\n            image_paths.extend([os.path.join(series_dir, img) for img in images])\n    return image_paths\n\n# Generate image paths for train and test data\ntrain_image_paths = generate_image_paths(train_desc, os.path.join(train_path, 'train_images'))\ntest_image_paths = generate_image_paths(test_desc, os.path.join(train_path, 'test_images'))\nprint(\"Example train image path:\", train_image_paths[2])\nprint(\"Total train series:\", len(train_desc))\nprint(\"Total train images found:\", len(train_image_paths))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:34:37.519627Z","iopub.execute_input":"2025-05-25T05:34:37.520576Z","iopub.status.idle":"2025-05-25T05:36:10.660630Z","shell.execute_reply.started":"2025-05-25T05:34:37.520546Z","shell.execute_reply":"2025-05-25T05:36:10.659524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to reshape a single row of the train CSV\ndef reshape_row(row):\n    data = {'study_id': [], 'condition': [], 'level': [], 'severity': []}\n    for column, value in row.items():\n        if column not in ['study_id', 'series_id', 'instance_number', 'x', 'y', 'series_description']:\n            parts = column.split('_')\n            condition = ' '.join([word.capitalize() for word in parts[:-2]])\n            level = parts[-2].capitalize() + '/' + parts[-1].capitalize()\n            data['study_id'].append(row['study_id'])\n            data['condition'].append(condition)\n            data['level'].append(level)\n            data['severity'].append(value)\n    return pd.DataFrame(data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:10.661672Z","iopub.execute_input":"2025-05-25T05:36:10.662051Z","iopub.status.idle":"2025-05-25T05:36:10.669043Z","shell.execute_reply.started":"2025-05-25T05:36:10.661988Z","shell.execute_reply":"2025-05-25T05:36:10.668204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reshape train CSV from wide to long format\nnew_train_df = pd.concat([reshape_row(row) for _, row in train.iterrows()], ignore_index=True)\nprint(\"Reshaped train data:\")\nprint(new_train_df.head(5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:10.671103Z","iopub.execute_input":"2025-05-25T05:36:10.671418Z","iopub.status.idle":"2025-05-25T05:36:11.867556Z","shell.execute_reply.started":"2025-05-25T05:36:10.671388Z","shell.execute_reply":"2025-05-25T05:36:11.866679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print column names for clarity\nprint(\"\\nColumns in new_train_df:\", \",\".join(new_train_df.columns))\nprint(\"Columns in label:\", \",\".join(label.columns))\nprint(\"Columns in test_desc:\", \",\".join(test_desc.columns))\nprint(\"Columns in sub:\", \",\".join(sub.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:11.868539Z","iopub.execute_input":"2025-05-25T05:36:11.868862Z","iopub.status.idle":"2025-05-25T05:36:11.875191Z","shell.execute_reply.started":"2025-05-25T05:36:11.868832Z","shell.execute_reply":"2025-05-25T05:36:11.874044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge dataframes: new_train_df, label and train_desc\nmerged_df = pd.merge(new_train_df, label, on=['study_id', 'condition', 'level'], how='inner')\nfinal_merged_df = pd.merge(merged_df, train_desc, on=['series_id', 'study_id'], how='inner')\nprint(\"Merged data sample:\")\nprint(final_merged_df.head(5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:11.876348Z","iopub.execute_input":"2025-05-25T05:36:11.876660Z","iopub.status.idle":"2025-05-25T05:36:11.971107Z","shell.execute_reply.started":"2025-05-25T05:36:11.876629Z","shell.execute_reply":"2025-05-25T05:36:11.970026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create new columns: row_id and image_path\nfinal_merged_df['row_id'] = (final_merged_df['study_id'].astype(str) + '_' +\n                             final_merged_df['condition'].str.lower().str.replace(' ', '_') + '_' +\n                             final_merged_df['level'].str.lower().str.replace('/', '_'))\nfinal_merged_df['image_path'] = (os.path.join(train_path, 'train_images') + '/' +\n                                 final_merged_df['study_id'].astype(str) + '/' +\n                                 final_merged_df['series_id'].astype(str) + '/' +\n                                 final_merged_df['instance_number'].astype(str) + '.dcm')\nprint(\"Data with new columns:\")\nprint(final_merged_df.head(5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:11.971969Z","iopub.execute_input":"2025-05-25T05:36:11.972438Z","iopub.status.idle":"2025-05-25T05:36:12.159054Z","shell.execute_reply.started":"2025-05-25T05:36:11.972408Z","shell.execute_reply":"2025-05-25T05:36:12.157820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map severity labels to lower-case format\nfinal_merged_df['severity'] = final_merged_df['severity'].map({'Normal/Mild': 'normal_mild',\n                                                               'Moderate': 'moderate',\n                                                               'Severe': 'severe'})\n\n# Set train_data as final_merged_df copy\ntrain_data = final_merged_df.copy()\nprint(\"Train data shape before filtering:\", train_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:12.160070Z","iopub.execute_input":"2025-05-25T05:36:12.160417Z","iopub.status.idle":"2025-05-25T05:36:12.194968Z","shell.execute_reply.started":"2025-05-25T05:36:12.160388Z","shell.execute_reply":"2025-05-25T05:36:12.193839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1) copy & strip out any stray whitespace\ndf = final_merged_df.copy()\ndf['series_description'] = df['series_description'].astype(str).str.strip()\n\n# 2) look at the unique views you actually have\nprint(\"ALL views in the data →\", df['series_description'].unique())\n\n# 3) build the counts pivot\ncounts = (\n    df\n    .groupby(['series_description','condition'])\n    .size()                     # count rows\n    .unstack(fill_value=0)      # make a DataFrame: rows=view, cols=condition\n)\n\n# 4) reorder to the three you care about (fill missing with 0)\ndesired = ['Sagittal T2/STIR','Sagittal T1','Axial T2']\ncounts = counts.reindex(desired, fill_value=0)\n\n# 5) see it\nprint(counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:12.196136Z","iopub.execute_input":"2025-05-25T05:36:12.196489Z","iopub.status.idle":"2025-05-25T05:36:12.264349Z","shell.execute_reply.started":"2025-05-25T05:36:12.196460Z","shell.execute_reply":"2025-05-25T05:36:12.263358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = final_merged_df.copy()\n\n# clean up the keys (optional)\ndf['series_description'] = df['series_description'].str.strip()\ndf['severity']           = df['severity'].str.lower().str.replace('/','_')  # e.g. \"Normal/Mild\"→\"normal_mild\"\n\n# now group by view → condition → severity, and count\ncond_sev_counts = (\n    df\n    .groupby(['series_description','condition','severity'])\n    .size()  # count rows\n    .unstack(fill_value=0)  # pivot severity → columns\n)\n\nprint(cond_sev_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:12.266425Z","iopub.execute_input":"2025-05-25T05:36:12.266698Z","iopub.status.idle":"2025-05-25T05:36:12.346349Z","shell.execute_reply.started":"2025-05-25T05:36:12.266676Z","shell.execute_reply":"2025-05-25T05:36:12.345260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 1) prep a “long” df\ndf = final_merged_df.copy()\ndf['series_description'] = df['series_description'].str.strip()\ndf['severity'] = (df['severity']\n                   .str.lower()\n                   .str.replace('/','_'))    # normal_mild, moderate, severe\n\n# 2) draw a separate count‐bar chart for each series_description\ng = sns.catplot(\n    data=df,\n    x='condition',\n    hue='severity',\n    col='series_description',\n    kind='count',\n    palette='muted',\n    height=4,\n    aspect=1.2,\n    sharey=False            # let each panel scale independently\n)\n\n# 3) polish\ng.set_axis_labels(\"\", \"Count\")\ng.set_titles(\"{col_name}\")  # just show the view name\nfor ax in g.axes.flat:\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=45, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:12.347305Z","iopub.execute_input":"2025-05-25T05:36:12.347538Z","iopub.status.idle":"2025-05-25T05:36:13.689954Z","shell.execute_reply.started":"2025-05-25T05:36:12.347520Z","shell.execute_reply":"2025-05-25T05:36:13.688968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ——————————————————————————————\n# 1) Total label‐rows per condition\nlabel_counts = (\n    final_merged_df\n    .groupby('condition')\n    .size()\n    .sort_values(ascending=False)\n    .rename(\"num_labels\")\n)\n\n# 2) Total unique images per condition\nimage_counts = (\n    final_merged_df\n    .groupby('condition')['image_path']\n    .nunique()\n    .sort_values(ascending=False)\n    .rename(\"num_images\")\n)\n\n# Combine into one DataFrame\ncond_summary = pd.concat([label_counts, image_counts], axis=1)\nprint(cond_summary)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T05:36:13.691141Z","iopub.execute_input":"2025-05-25T05:36:13.691428Z","iopub.status.idle":"2025-05-25T05:36:13.738335Z","shell.execute_reply.started":"2025-05-25T05:36:13.691406Z","shell.execute_reply":"2025-05-25T05:36:13.737114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# plot num_images per condition\ncond_summary['num_images'].plot.barh(figsize=(8, 5))\nplt.title(\"Number of Unique Images per Condition\")\nplt.xlabel(\"Unique DICOM Files\")\nplt.ylabel(\"Condition\")\nplt.gca().invert_yaxis()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T06:07:19.700218Z","iopub.execute_input":"2025-05-25T06:07:19.700550Z","iopub.status.idle":"2025-05-25T06:07:19.952689Z","shell.execute_reply.started":"2025-05-25T06:07:19.700527Z","shell.execute_reply":"2025-05-25T06:07:19.951630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# train_desc      : the DataFrame of all series in train\n# final_merged_df : the DataFrame where each row is one (slice,condition,level) annotation\n\nfrom pathlib import Path\n\n# 1) total files on disk under train_images/\ntrain_images_dir = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images')\nall_paths = list(train_images_dir.rglob('*.dcm'))\nprint(\"Total DICOM files in train:\", len(all_paths))\n\n# 2) how many of those were actually annotated at least once?\n# final_merged_df['image_path'] should hold the full path to each labelled slice\nn_labelled = final_merged_df['image_path'].nunique()\nprint(\"Unique images with ≥1 label:\", n_labelled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T06:08:45.291050Z","iopub.execute_input":"2025-05-25T06:08:45.291409Z","iopub.status.idle":"2025-05-25T06:10:34.944523Z","shell.execute_reply.started":"2025-05-25T06:08:45.291385Z","shell.execute_reply":"2025-05-25T06:10:34.943282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}