{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport random\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nimport polars as pl\npl.Config.set_tbl_rows(40)\npl.Config.set_fmt_str_lengths(n=40)\nimport seaborn as sns\nimport pydicom\nimport math\nimport missingno as msno\nfrom skimage.draw import disk\nfrom skimage.io import imsave\nfrom zipfile import ZipFile\n\n\ntqdm.pandas()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:32:58.372295Z","iopub.execute_input":"2024-09-06T00:32:58.372678Z","iopub.status.idle":"2024-09-06T00:32:58.380277Z","shell.execute_reply.started":"2024-09-06T00:32:58.372644Z","shell.execute_reply":"2024-09-06T00:32:58.379241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:blue\"> This is my second time of documenting, so any feedbacks are appreciated!!!</span>","metadata":{}},{"cell_type":"markdown","source":"# Objective\n1. Handling data inbalances in each csv file.\n2. Image preprocessing\n3. Improving the image mask with weights (coming soon!)\n\nMy previous EDA notebook: https://www.kaggle.com/code/toyakki/rsna24-eda-preprocess?scriptVersionId=188151583","metadata":{}},{"cell_type":"code","source":"rd = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification'\ntrain = pd.read_csv(os.path.join(rd, \"train.csv\"))\ntrain_labels = pd.read_csv(os.path.join(rd, \"train_label_coordinates.csv\"))\ntrain_des = pd.read_csv(os.path.join(rd, \"train_series_descriptions.csv\"))\ntest_des = pd.read_csv(os.path.join(rd, \"test_series_descriptions.csv\"))\nsample_sub = pd.read_csv(os.path.join(rd, \"sample_submission.csv\"))","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:32:59.447039Z","iopub.execute_input":"2024-09-06T00:32:59.447387Z","iopub.status.idle":"2024-09-06T00:32:59.546246Z","shell.execute_reply.started":"2024-09-06T00:32:59.447363Z","shell.execute_reply":"2024-09-06T00:32:59.545275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Filling NaN values\nAlthough the null values may not matter that much when coming to image preprocessing, it is a good practice to replace nan values with the categorical labels. I assume about the two facts from EDA:\n1. The severity of degeneration is identical on both the left and right sides of the spine at the same vertebral level.\n\n2. If there are no images from a specific region of the spine are not detected, then the patient does not exhibit any degenerative issues in that region. For example, the following cases do not have any images of foramen which diagonoses neural foraminal narrowing, so they will be categorized as 'Normal/Mild'.","metadata":{}},{"cell_type":"code","source":"train[train['left_neural_foraminal_narrowing_l1_l2'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:32:59.698447Z","iopub.execute_input":"2024-09-06T00:32:59.698739Z","iopub.status.idle":"2024-09-06T00:32:59.735805Z","shell.execute_reply.started":"2024-09-06T00:32:59.698708Z","shell.execute_reply":"2024-09-06T00:32:59.734928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Before","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:00.480452Z","iopub.execute_input":"2024-09-06T00:33:00.480788Z","iopub.status.idle":"2024-09-06T00:33:00.495792Z","shell.execute_reply.started":"2024-09-06T00:33:00.480763Z","shell.execute_reply":"2024-09-06T00:33:00.494703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train[train.filter(like='spinal_canal').columns].isnull().any(axis=1)]","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:00.630501Z","iopub.execute_input":"2024-09-06T00:33:00.630757Z","iopub.status.idle":"2024-09-06T00:33:00.653532Z","shell.execute_reply.started":"2024-09-06T00:33:00.630735Z","shell.execute_reply":"2024-09-06T00:33:00.652546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Implementing the assumption 2","metadata":{}},{"cell_type":"code","source":"spinal_canal_columns = train.filter(like='spinal_canal').columns\ntrain[spinal_canal_columns] = train[spinal_canal_columns].fillna('Normal/Mild')","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:00.897106Z","iopub.execute_input":"2024-09-06T00:33:00.897369Z","iopub.status.idle":"2024-09-06T00:33:00.906825Z","shell.execute_reply.started":"2024-09-06T00:33:00.897347Z","shell.execute_reply":"2024-09-06T00:33:00.906040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Implementing the assumption 1","metadata":{}},{"cell_type":"code","source":"def fill_nan_left_and_right(df, column):\n    target_columns = df.filter(like=column).columns \n    null_rows = df[df[target_columns].isnull().any(axis=1)]\n    for level in ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']:\n        left_col = f\"left_{column}_{level}\"\n        right_col = f\"right_{column}_{level}\"\n        for index, row in null_rows.iterrows():\n            if pd.isna(row[left_col]) and pd.isna(row[right_col]):\n                df.at[index, left_col] = 'Normal/Mild'\n                df.at[index, right_col] = 'Normal/Mild'\n            \n            elif pd.isna(row[right_col]):\n                df.at[index, right_col] = row[left_col]\n            \n            elif pd.isna(row[left_col]):\n                df.at[index, left_col] = row[right_col]\n            else:\n                continue\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:01.097911Z","iopub.execute_input":"2024-09-06T00:33:01.098173Z","iopub.status.idle":"2024-09-06T00:33:01.105461Z","shell.execute_reply.started":"2024-09-06T00:33:01.098150Z","shell.execute_reply":"2024-09-06T00:33:01.104554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = fill_nan_left_and_right(train, 'neural_foraminal_narrowing')\ntrain = fill_nan_left_and_right(train, 'subarticular_stenosis')","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:01.225458Z","iopub.execute_input":"2024-09-06T00:33:01.225708Z","iopub.status.idle":"2024-09-06T00:33:01.319356Z","shell.execute_reply.started":"2024-09-06T00:33:01.225687Z","shell.execute_reply":"2024-09-06T00:33:01.318661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## After","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:01.488932Z","iopub.execute_input":"2024-09-06T00:33:01.489193Z","iopub.status.idle":"2024-09-06T00:33:01.501910Z","shell.execute_reply.started":"2024-09-06T00:33:01.489170Z","shell.execute_reply":"2024-09-06T00:33:01.501162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[\"series_description\"] = train_labels.progress_apply(\nlambda row: train_des.loc[(train_des.study_id == row.study_id) & (train_des.series_id == row.series_id)]\n[\"series_description\"].tolist()[0], axis=1)\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:01.702223Z","iopub.execute_input":"2024-09-06T00:33:01.702498Z","iopub.status.idle":"2024-09-06T00:33:29.282301Z","shell.execute_reply.started":"2024-09-06T00:33:01.702473Z","shell.execute_reply":"2024-09-06T00:33:29.281484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example picture to preprocess\ndcm_path = os.path.join(rd,'train_images/4003253/702807833/10.dcm')\ndicom_data = pydicom.dcmread(dcm_path)\nimage_shape = dicom_data.pixel_array.shape\nprint(f\"Image size (in pixels): {image_shape}\")\n\n\nif 'PixelSpacing' in dicom_data:\n    pixel_spacing = dicom_data.PixelSpacing\n    print(f\"Pixel spacing (in mm): {pixel_spacing}\")\n\n\nif 'SliceThickness' in dicom_data:\n    slice_thickness = dicom_data.SliceThickness\n    print(f\"Slice thickness (in mm): {slice_thickness}\")\n\nif 'SpacingBetweenSlices' in dicom_data:\n    spacing_between_slices = dicom_data.SpacingBetweenSlices\n    print(f\"Spacing between slices (in mm): {spacing_between_slices}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:33:29.284090Z","iopub.execute_input":"2024-09-06T00:33:29.284380Z","iopub.status.idle":"2024-09-06T00:33:29.327582Z","shell.execute_reply.started":"2024-09-06T00:33:29.284354Z","shell.execute_reply":"2024-09-06T00:33:29.326782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DICOMPreprocessor class\n\n## Steps:\n1. Read the DICOM file.\n2. Resampling the image: Resampling to the certain image resolution to remove variance.\n3. Resize the image: Sale the image dimensions to a fixed size, such as 512 times 512 pixels.\n4. Normalizing pixel values: Scale pixel intensity values to a range suitable for our model.\n\n## Potential changes you can make:\n1. Using the scipy.ndimage.zoom for resampling the images. That method is introduced here: https://www.kaggle.com/code/gzuidhof/full-preprocessing-tutorial\n\n2. Handling different slice thickness.\n3. More preprocessing step in the class.","metadata":{}},{"cell_type":"code","source":"import logging\nfrom skimage.transform import resize\nfrom skimage.io import imsave\n\n\nclass DICOMPreprocessor:\n    def __init__(self, target_spacing=(1.0, 1.0), target_size=(512, 512)):\n        self.target_spacing = target_spacing\n        self.target_size = target_size\n        logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')\n        self.logger = logging.getLogger(__name__)\n        \n    def load_dicom(self, path):\n        dicom = pydicom.dcmread(path)\n        image = dicom.pixel_array\n        return dicom, image\n    \n    def resample_image(self, image, dicom):\n        if 'PixelSpacing' in dicom:\n            current_spacing = np.array(dicom.PixelSpacing, dtype=np.float32)\n            resize_factor = current_spacing/ self.target_spacing\n            new_shape = np.round(image.shape * resize_factor).astype(int)\n            resampled_image = resize(image, new_shape, preserve_range=True, anti_aliasing=True)\n            return resampled_image\n        \n        else:\n            self.logger.error(f\"PixelSpacing not found in DICOM metadata for {dicom.filename}. Skipping resampling.\")\n            return image\n        \n    def resize_image(self, image):\n        resized_image = resize(image, self.target_size, preserve_range=True, anti_aliasing=True)\n        return resized_image\n    \n    def normalize_image(self, image):\n        normalized_image = (image - np.min(image)) / (np.max(image) - np.min(image))\n        return normalized_image\n    \n    def preprocess(self, dicom_path):\n        try:\n            dicom, image = self.load_dicom(dicom_path)\n            resampled_image = self.resample_image(image, dicom)\n            resized_image = self.resize_image(resampled_image)\n            normalized_image = self.normalize_image(resized_image)\n            return normalized_image\n        except Exception as e:\n            self.logger.error(f\"Error processing file {dicom_path}: {e}\")\n            return None\n    \n    def save_as_png(self, image, save_path):\n        imsave(save_path, (image * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:26.974574Z","iopub.execute_input":"2024-09-06T00:37:26.975188Z","iopub.status.idle":"2024-09-06T00:37:27.012706Z","shell.execute_reply.started":"2024-09-06T00:37:26.975157Z","shell.execute_reply":"2024-09-06T00:37:27.011878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = DICOMPreprocessor()","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:55.114416Z","iopub.execute_input":"2024-09-06T00:37:55.115492Z","iopub.status.idle":"2024-09-06T00:37:55.119517Z","shell.execute_reply.started":"2024-09-06T00:37:55.115448Z","shell.execute_reply":"2024-09-06T00:37:55.118504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert into png format and save","metadata":{}},{"cell_type":"code","source":"import re\n\nst_ids = train_labels['study_id'].unique()\ndesc = train_labels['series_description'].unique()\n\ndef atoi(text):\n    return int(text) if text.isdigit() else text\n\ndef natural_keys(text):\n    return [atoi(c) for c in re.split(r'(\\d+)', text) ]","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:56.887298Z","iopub.execute_input":"2024-09-06T00:37:56.887959Z","iopub.status.idle":"2024-09-06T00:37:56.897471Z","shell.execute_reply.started":"2024-09-06T00:37:56.887925Z","shell.execute_reply":"2024-09-06T00:37:56.896773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:57.051972Z","iopub.execute_input":"2024-09-06T00:37:57.052452Z","iopub.status.idle":"2024-09-06T00:37:57.066397Z","shell.execute_reply.started":"2024-09-06T00:37:57.052427Z","shell.execute_reply":"2024-09-06T00:37:57.065555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_des['series_description'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:57.425608Z","iopub.execute_input":"2024-09-06T00:37:57.425919Z","iopub.status.idle":"2024-09-06T00:37:57.431600Z","shell.execute_reply.started":"2024-09-06T00:37:57.425887Z","shell.execute_reply":"2024-09-06T00:37:57.430781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport glob\n\nst_ids = train_labels['study_id'].unique()\ndesc = train_labels['series_description'].unique()\n\nfor idx, s_id in enumerate(tqdm(st_ids, total=len(st_ids))):\n    pdf = train_labels[train_labels['study_id'] == s_id]\n    for ds in desc:\n        ds_ = ds.replace('/', '_')\n        pdf_ = pdf[pdf['series_description'] == ds]\n        output_dir = f'cvt_png/{s_id}/{ds_}'\n        os.makedirs(output_dir, exist_ok=True)\n        \n        all_imgs = [img for i, row in pdf_.iterrows() \n                   for img in sorted(glob.glob(f'{rd}/train_images/{row[\"study_id\"]}/{row[\"series_id\"]}/*.dcm'), \n                                     key=natural_keys)][:10]\n        \n        \n        for j, dicom_path in enumerate(all_imgs):\n            if j >= 10: \n                break\n            preprocessed_image = preprocessor.preprocess(dicom_path)\n            if preprocessed_image is not None:\n                save_path = f'{output_dir}/{j:03d}.png'\n                preprocessor.save_as_png(preprocessed_image, save_path)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T00:37:58.071974Z","iopub.execute_input":"2024-09-06T00:37:58.072295Z","iopub.status.idle":"2024-09-06T02:02:29.936977Z","shell.execute_reply.started":"2024-09-06T00:37:58.072271Z","shell.execute_reply":"2024-09-06T02:02:29.935924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_id = test_des['study_id'].unique()[0]\nfor ds in desc:\n    ds_ = ds.replace('/', '_')\n    df = test_des[test_des['series_description'] == ds]\n    output_dir_test = f'cvt_png/test/{s_id}/{ds_}'\n    os.makedirs(output_dir_test, exist_ok=True)\n    \n    all_imgs = sorted([f'{rd}/test_images/{row[\"study_id\"]}/{row[\"series_id\"]}/*.dcm'\n                              for i, row in df.iterrows()], key=natural_keys)\n    \n    for j, dicom_path in enumerate(all_imgs):\n            if j >= 10: \n                break\n            preprocessed_image = preprocessor.preprocess(dicom_path)\n            if preprocessed_image is not None:\n                save_path = f'{output_dir_test}/{j:03d}.png'\n                preprocessor.save_as_png(preprocessed_image, save_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}