{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport random\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nimport polars as pl\npl.Config.set_tbl_rows(40)\npl.Config.set_fmt_str_lengths(n=40)\nimport seaborn as sns\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-05T05:46:45.684343Z","iopub.execute_input":"2024-07-05T05:46:45.685754Z","iopub.status.idle":"2024-07-05T05:46:52.018573Z","shell.execute_reply.started":"2024-07-05T05:46:45.685665Z","shell.execute_reply":"2024-07-05T05:46:52.016661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.026992Z","iopub.execute_input":"2024-07-05T05:46:52.027984Z","iopub.status.idle":"2024-07-05T05:46:52.040531Z","shell.execute_reply.started":"2024-07-05T05:46:52.027899Z","shell.execute_reply":"2024-07-05T05:46:52.038064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\ntrain = pd.read_csv(os.path.join(root_dir, \"train.csv\"))\ntrain_labels = pd.read_csv(os.path.join(root_dir, \"train_label_coordinates.csv\"))\ntrain_des = pd.read_csv(os.path.join(root_dir, \"train_series_descriptions.csv\"))\ntest_des = pd.read_csv(os.path.join(root_dir, \"test_series_descriptions.csv\"))\nsample_sub = pd.read_csv(os.path.join(root_dir, \"sample_submission.csv\"))","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.042852Z","iopub.execute_input":"2024-07-05T05:46:52.043603Z","iopub.status.idle":"2024-07-05T05:46:52.198021Z","shell.execute_reply.started":"2024-07-05T05:46:52.043546Z","shell.execute_reply":"2024-07-05T05:46:52.196547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.shape, train_des.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.201556Z","iopub.execute_input":"2024-07-05T05:46:52.202016Z","iopub.status.idle":"2024-07-05T05:46:52.212435Z","shell.execute_reply.started":"2024-07-05T05:46:52.201980Z","shell.execute_reply":"2024-07-05T05:46:52.210945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.214416Z","iopub.execute_input":"2024-07-05T05:46:52.214879Z","iopub.status.idle":"2024-07-05T05:46:52.244746Z","shell.execute_reply.started":"2024-07-05T05:46:52.214839Z","shell.execute_reply":"2024-07-05T05:46:52.243232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_des.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.247129Z","iopub.execute_input":"2024-07-05T05:46:52.247736Z","iopub.status.idle":"2024-07-05T05:46:52.268966Z","shell.execute_reply.started":"2024-07-05T05:46:52.247683Z","shell.execute_reply":"2024-07-05T05:46:52.267467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Distribution Analysis","metadata":{}},{"cell_type":"markdown","source":"Adding `series description` from the `train_des` dataframe. Like, merging `train_labels` and `train_des` dataframe by `study_id` and `series_id`.","metadata":{}},{"cell_type":"code","source":"train_labels[\"series_description\"] = train_labels.progress_apply(lambda row: train_des.loc[(train_des.study_id == row.study_id) & (train_des.series_id == row.series_id)][\"series_description\"].tolist()[0], axis=1)\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:46:52.271076Z","iopub.execute_input":"2024-07-05T05:46:52.271651Z","iopub.status.idle":"2024-07-05T05:47:29.790796Z","shell.execute_reply.started":"2024-07-05T05:46:52.271573Z","shell.execute_reply":"2024-07-05T05:47:29.788983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Instances per `series_description` in `train_labels`.","metadata":{}},{"cell_type":"code","source":"train_labels[[\"series_description\", \"instance_number\"]].groupby([\"series_description\"]).count()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:29.792687Z","iopub.execute_input":"2024-07-05T05:47:29.793120Z","iopub.status.idle":"2024-07-05T05:47:29.827548Z","shell.execute_reply.started":"2024-07-05T05:47:29.793068Z","shell.execute_reply":"2024-07-05T05:47:29.825003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Instances per `series_description` and `condition` in `train_labels` \n\nIt shows that there is a miss leveled in condition in Sagittal T1. As per my understanding `Spinal Canal Stenosis` can only be in `Sagittal T2/STIR`.","metadata":{}},{"cell_type":"code","source":"train_labels[[\"series_description\", \"condition\", \"instance_number\"]].groupby([\"series_description\", \"condition\"]).count()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:29.829815Z","iopub.execute_input":"2024-07-05T05:47:29.830460Z","iopub.status.idle":"2024-07-05T05:47:29.874042Z","shell.execute_reply.started":"2024-07-05T05:47:29.830402Z","shell.execute_reply":"2024-07-05T05:47:29.871804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Resolving the issues found earlier","metadata":{}},{"cell_type":"code","source":"train_labels.loc[(train_labels.series_description == \"Sagittal T1\") & (train_labels.condition == \"Spinal Canal Stenosis\"), \"series_description\"] = \"Sagittal T2/STIR\"","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:29.876245Z","iopub.execute_input":"2024-07-05T05:47:29.877286Z","iopub.status.idle":"2024-07-05T05:47:29.915097Z","shell.execute_reply.started":"2024-07-05T05:47:29.877213Z","shell.execute_reply":"2024-07-05T05:47:29.912714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[[\"series_description\", \"condition\", \"instance_number\"]].groupby([\"series_description\", \"condition\"]).count()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:29.917803Z","iopub.execute_input":"2024-07-05T05:47:29.918426Z","iopub.status.idle":"2024-07-05T05:47:29.970952Z","shell.execute_reply.started":"2024-07-05T05:47:29.918374Z","shell.execute_reply":"2024-07-05T05:47:29.969268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Instances per `series_description`, `condition` and `level` in `train_labels` ","metadata":{}},{"cell_type":"code","source":"dist = train_labels[[\"series_description\", \"condition\", \"level\", \"instance_number\"]].groupby([\"series_description\", \"condition\", \"level\"]).count()\ndist","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:29.973324Z","iopub.execute_input":"2024-07-05T05:47:29.973866Z","iopub.status.idle":"2024-07-05T05:47:30.029233Z","shell.execute_reply.started":"2024-07-05T05:47:29.973821Z","shell.execute_reply":"2024-07-05T05:47:30.026559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Annotation Analysis","metadata":{}},{"cell_type":"code","source":"# study_id = train.iloc[random.randint(0, 1000)].study_id\nstudy_id = 901299313","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.035864Z","iopub.execute_input":"2024-07-05T05:47:30.036471Z","iopub.status.idle":"2024-07-05T05:47:30.045789Z","shell.execute_reply.started":"2024-07-05T05:47:30.036420Z","shell.execute_reply":"2024-07-05T05:47:30.043575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.loc[train[\"study_id\"] == study_id]","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.047929Z","iopub.execute_input":"2024-07-05T05:47:30.049042Z","iopub.status.idle":"2024-07-05T05:47:30.094231Z","shell.execute_reply.started":"2024-07-05T05:47:30.048985Z","shell.execute_reply":"2024-07-05T05:47:30.092491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study = train_labels.loc[train_labels[\"study_id\"] == study_id]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.096663Z","iopub.execute_input":"2024-07-05T05:47:30.097364Z","iopub.status.idle":"2024-07-05T05:47:30.131462Z","shell.execute_reply.started":"2024-07-05T05:47:30.097289Z","shell.execute_reply":"2024-07-05T05:47:30.129471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_desc_study = train_des.loc[train_des[\"study_id\"] == study_id]\ntrain_desc_study","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.133960Z","iopub.execute_input":"2024-07-05T05:47:30.134696Z","iopub.status.idle":"2024-07-05T05:47:30.155086Z","shell.execute_reply.started":"2024-07-05T05:47:30.134639Z","shell.execute_reply":"2024-07-05T05:47:30.153172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding `series description` from the `train_des` dataframe. Like, merging `train_labels` and `train_des` dataframe by `study_id` and `series_id`.","metadata":{}},{"cell_type":"code","source":"for i, row in train_desc_study.iterrows():\n    train_study.loc[train_study.series_id == row.series_id, \"series_description\"] = row.series_description\n\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.157769Z","iopub.execute_input":"2024-07-05T05:47:30.158854Z","iopub.status.idle":"2024-07-05T05:47:30.194084Z","shell.execute_reply.started":"2024-07-05T05:47:30.158772Z","shell.execute_reply":"2024-07-05T05:47:30.192528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding `position (x, y, z)` from the `train_des` dataframe.","metadata":{}},{"cell_type":"code","source":"def image_position(ints):\n    sid = ints.study_id\n    series_id = ints.series_id\n    instance_id = ints.instance_number\n\n    dicom = pydicom.read_file(root_dir + f\"/train_images/{sid}/{series_id}/{instance_id}.dcm\")\n    return dicom.ImagePositionPatient\n\ntrain_study.loc[:, [\"px\", \"py\", \"pz\"]] = train_study.apply(lambda x: image_position(x), axis=1).tolist()\ntrain_study.sort_values(by=[\"series_description\", \"instance_number\"])","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.196045Z","iopub.execute_input":"2024-07-05T05:47:30.197432Z","iopub.status.idle":"2024-07-05T05:47:30.305184Z","shell.execute_reply.started":"2024-07-05T05:47:30.197385Z","shell.execute_reply":"2024-07-05T05:47:30.303830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\ndef load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef plot_image(series_description, columns = 3, size=[16, 6]):\n    ints = train_study.loc[train_study[\"series_description\"] == series_description]\n    instance_number = ints.instance_number.unique()\n    \n    rows = math.floor(len(instance_number) / columns) + 1\n    size[1] = size[1] * rows\n    fig = plt.figure(figsize=size)\n    for i, i_n in enumerate(instance_number):\n        ints_df = ints.loc[ints.instance_number == i_n]\n        sid = ints_df.iloc[0].study_id\n        series_id = ints_df.iloc[0].series_id\n        instance_id = ints_df.iloc[0].instance_number\n        try:\n            image = load_dicom(root_dir + f\"/train_images/{sid}/{series_id}/{instance_id}.dcm\")\n            ax = fig.add_subplot(rows, columns, i+1)\n            plt.imshow(image, cmap=\"gray\")\n\n            for i, row in ints_df.iterrows():\n                circle = plt.Circle((row[\"x\"], row[\"y\"]), 10, color='r', fill=False)\n                ax.add_patch(circle)\n        except:\n            continue\n            \n    plt.show()\n    print(ints.drop_duplicates([\"instance_number\"]).reset_index(drop=True).to_string())","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.307090Z","iopub.execute_input":"2024-07-05T05:47:30.307623Z","iopub.status.idle":"2024-07-05T05:47:30.323025Z","shell.execute_reply.started":"2024-07-05T05:47:30.307581Z","shell.execute_reply":"2024-07-05T05:47:30.321438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Viewing annotations per images for Axial T2 Plane with showing the annotations df at the end","metadata":{}},{"cell_type":"code","source":"plot_image(\"Axial T2\", size=[16, 5])","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:30.324975Z","iopub.execute_input":"2024-07-05T05:47:30.325538Z","iopub.status.idle":"2024-07-05T05:47:32.829376Z","shell.execute_reply.started":"2024-07-05T05:47:30.325498Z","shell.execute_reply":"2024-07-05T05:47:32.826959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Viewing annotations per images for Sagittal T1 Plane with showing the annotations df at the end","metadata":{}},{"cell_type":"code","source":"plot_image(\"Sagittal T1\", size=[16, 3])","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:32.831909Z","iopub.execute_input":"2024-07-05T05:47:32.832617Z","iopub.status.idle":"2024-07-05T05:47:34.170234Z","shell.execute_reply.started":"2024-07-05T05:47:32.832561Z","shell.execute_reply":"2024-07-05T05:47:34.167791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Viewing annotations per images for Sagittal T2/STIR Plane with showing the annotations df at the end","metadata":{}},{"cell_type":"code","source":"plot_image(\"Sagittal T2/STIR\", size=[16, 3])","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:34.172998Z","iopub.execute_input":"2024-07-05T05:47:34.173638Z","iopub.status.idle":"2024-07-05T05:47:34.972983Z","shell.execute_reply.started":"2024-07-05T05:47:34.173591Z","shell.execute_reply":"2024-07-05T05:47:34.970869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KFold Technic","metadata":{}},{"cell_type":"markdown","source":"In this KFold technic, we have selected the study ids as the source of the dataset. Because, we want to train a model with a dataset and validate the model with a dataset that is unknown to the model, meaning that was not included in the training set.\n\nIn this technic I found that we can ensure that perfectly. Because we are training with a dataset containing data for some study ids and validation set contains data for some other study ids.","metadata":{}},{"cell_type":"code","source":"study_ids = np.array(train_labels.study_id.unique())\nlen(study_ids)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:34.975568Z","iopub.execute_input":"2024-07-05T05:47:34.976353Z","iopub.status.idle":"2024-07-05T05:47:34.990409Z","shell.execute_reply.started":"2024-07-05T05:47:34.976290Z","shell.execute_reply":"2024-07-05T05:47:34.988878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nfor fold, (trn_idx, val_idx) in enumerate(kf.split(range(len(study_ids)))):\n    print(f\"Fold {fold}\")\n    print(\"Train studies:\", len(trn_idx))\n    trn_study_id = study_ids[trn_idx]\n    fold_train_labels = train_labels.loc[train_labels.study_id.isin(trn_study_id)].reset_index(drop=True)\n    fold_train_des = train_des.loc[train_des.study_id.isin(trn_study_id)].reset_index(drop=True)\n    print(\"Fold train labels shape:\", fold_train_labels.shape, \"\\nFold train des shape:\", fold_train_des.shape)\n\n    print(\"Valid studies:\", len(val_idx))\n    val_study_id = study_ids[val_idx]\n    fold_valid_labels = train_labels.loc[train_labels.study_id.isin(val_study_id)].reset_index(drop=True)\n    fold_valid_des = train_des.loc[train_des.study_id.isin(val_study_id)].reset_index(drop=True)\n    print(\"Fold valid labels shape:\", fold_valid_labels.shape, \"\\nFold valid des shape:\", fold_valid_des.shape)\n    print(\"*\" * 40)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T05:47:34.992089Z","iopub.execute_input":"2024-07-05T05:47:34.992588Z","iopub.status.idle":"2024-07-05T05:47:35.141903Z","shell.execute_reply.started":"2024-07-05T05:47:34.992547Z","shell.execute_reply":"2024-07-05T05:47:35.137483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}