{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:40.522044Z","iopub.execute_input":"2024-07-20T06:40:40.522689Z","iopub.status.idle":"2024-07-20T06:40:40.550017Z","shell.execute_reply.started":"2024-07-20T06:40:40.522656Z","shell.execute_reply":"2024-07-20T06:40:40.549087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\n\nimport pandas as pd\nimport numpy as np\n\nfrom tqdm import tqdm\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport cv2\nfrom math import floor\n\nfrom joblib import delayed, Parallel\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-20T06:42:35.165201Z","iopub.execute_input":"2024-07-20T06:42:35.165559Z","iopub.status.idle":"2024-07-20T06:42:35.171271Z","shell.execute_reply.started":"2024-07-20T06:42:35.165533Z","shell.execute_reply":"2024-07-20T06:42:35.169962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir('volumes')","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:44:19.721567Z","iopub.execute_input":"2024-07-20T06:44:19.722663Z","iopub.status.idle":"2024-07-20T06:44:19.727772Z","shell.execute_reply.started":"2024-07-20T06:44:19.722593Z","shell.execute_reply":"2024-07-20T06:44:19.726577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utils","metadata":{}},{"cell_type":"code","source":"class Config():\n    \n    n_slices_per_series = 10\n    image_size = (256, 256)\n    interpolation = cv2.INTER_CUBIC\n    \n    root = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification'","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.082566Z","iopub.execute_input":"2024-07-20T06:40:42.083308Z","iopub.status.idle":"2024-07-20T06:40:42.088111Z","shell.execute_reply.started":"2024-07-20T06:40:42.083279Z","shell.execute_reply":"2024-07-20T06:40:42.087051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = os.path.join(Config.root, 'train_images')","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.089419Z","iopub.execute_input":"2024-07-20T06:40:42.089753Z","iopub.status.idle":"2024-07-20T06:40:42.098821Z","shell.execute_reply.started":"2024-07-20T06:40:42.089728Z","shell.execute_reply":"2024-07-20T06:40:42.097522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_types = ['Sagittal T2/STIR', 'Sagittal T1', 'Axial T2']","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.101009Z","iopub.execute_input":"2024-07-20T06:40:42.101370Z","iopub.status.idle":"2024-07-20T06:40:42.109191Z","shell.execute_reply.started":"2024-07-20T06:40:42.101343Z","shell.execute_reply":"2024-07-20T06:40:42.108200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def shift_image_bits(img, dcm):\n    \n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = img.dtype \n        img = (img << bit_shift).astype(dtype) >>  bit_shift\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.110419Z","iopub.execute_input":"2024-07-20T06:40:42.110759Z","iopub.status.idle":"2024-07-20T06:40:42.119644Z","shell.execute_reply.started":"2024-07-20T06:40:42.110732Z","shell.execute_reply":"2024-07-20T06:40:42.118576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_window(img, dcm):\n\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    # Some notebooks instead of clipping\n    # with window parameters use quantiles\n    img = np.clip(img, low, high)\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.120917Z","iopub.execute_input":"2024-07-20T06:40:42.121331Z","iopub.status.idle":"2024-07-20T06:40:42.129702Z","shell.execute_reply.started":"2024-07-20T06:40:42.121282Z","shell.execute_reply":"2024-07-20T06:40:42.128658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_dcm_image(image_folder, study_id, series_id, instance):\n    \n    dcm_path = os.path.join(Config.root,\n                            image_folder, \n                            str(study_id),\n                            str(series_id),\n                            f'{instance}.dcm')\n    dcm = pydicom.dcmread(dcm_path)\n    \n    img = dcm.pixel_array\n    img = shift_image_bits(img, dcm)\n    img = apply_window(img, dcm)\n    \n    # Resizing performed before image normalization!\n    img = cv2.resize(img, \n                 Config.image_size, \n                 interpolation=Config.interpolation)\n    \n    # Maybe we should normalize not slice,\n    # but the whole series of slices\n    if img.max() != img.min():\n        img = (img - img.min()) / (img.max() - img.min())\n    else:\n        img = img - img\n\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n        \n    img = (img * 255).astype('uint8')\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.130855Z","iopub.execute_input":"2024-07-20T06:40:42.131191Z","iopub.status.idle":"2024-07-20T06:40:42.139950Z","shell.execute_reply.started":"2024-07-20T06:40:42.131163Z","shell.execute_reply":"2024-07-20T06:40:42.138934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_instances_df(series_df):\n    \n    files_df = []\n\n    for study_id in series_df.study_id.unique():\n\n        study_filter = train_series.study_id == study_id\n        series_ids = train_series.loc[study_filter, 'series_id']\n\n        for series_id in series_ids:\n\n            series_filter = train_series.series_id == series_id\n            series_description = train_series.loc[series_filter, 'series_description'].iloc[0]\n\n            series_folder = os.path.join(train_path, str(study_id), str(series_id))\n            files = os.listdir(series_folder)\n            files = [int(file[:-4]) for file in files]\n\n            for file in files:\n                files_df.append((study_id, series_id, file, series_description))\n\n    files_columns = ['study_id', 'series_id', 'instance', 'series_type']\n    files_df = pd.DataFrame(files_df, columns=files_columns)\n\n    files_df = files_df.sort_values(['study_id', 'series_id', 'instance'], ascending=True)\n    \n    return files_df","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.141124Z","iopub.execute_input":"2024-07-20T06:40:42.141443Z","iopub.status.idle":"2024-07-20T06:40:42.154850Z","shell.execute_reply.started":"2024-07-20T06:40:42.141418Z","shell.execute_reply":"2024-07-20T06:40:42.153931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data ","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv(os.path.join(Config.root, 'train.csv'))\ntrain_series = pd.read_csv(os.path.join(Config.root, 'train_series_descriptions.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.427758Z","iopub.execute_input":"2024-07-20T06:40:42.428123Z","iopub.status.idle":"2024-07-20T06:40:42.472126Z","shell.execute_reply.started":"2024-07-20T06:40:42.428094Z","shell.execute_reply":"2024-07-20T06:40:42.471093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example of strange file enumeration.","metadata":{}},{"cell_type":"code","source":"files = os.listdir(os.path.join(train_path, '2581283971', '2683794967'))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:42.806794Z","iopub.execute_input":"2024-07-20T06:40:42.807756Z","iopub.status.idle":"2024-07-20T06:40:42.824654Z","shell.execute_reply.started":"2024-07-20T06:40:42.807723Z","shell.execute_reply":"2024-07-20T06:40:42.823662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted(files)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:43.017246Z","iopub.execute_input":"2024-07-20T06:40:43.017615Z","iopub.status.idle":"2024-07-20T06:40:43.024695Z","shell.execute_reply.started":"2024-07-20T06:40:43.017586Z","shell.execute_reply":"2024-07-20T06:40:43.023688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Count of slices doesn't equal to the max instance number.","metadata":{}},{"cell_type":"markdown","source":"### Series order","metadata":{}},{"cell_type":"code","source":"train_series[train_series.study_id == 10728036]","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:43.518369Z","iopub.execute_input":"2024-07-20T06:40:43.519241Z","iopub.status.idle":"2024-07-20T06:40:43.540403Z","shell.execute_reply.started":"2024-07-20T06:40:43.519209Z","shell.execute_reply":"2024-07-20T06:40:43.539462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When patient has multiple series with the same description we will concat them using the same order as in dataframe, i.e. monotonic.","metadata":{}},{"cell_type":"markdown","source":"## All instances for each series","metadata":{}},{"cell_type":"code","source":"files_df = get_all_instances_df(train_series)\n\nprint(f'Shape of files_df - {files_df.shape}')\n\nfiles_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:40:44.033551Z","iopub.execute_input":"2024-07-20T06:40:44.033933Z","iopub.status.idle":"2024-07-20T06:41:44.143168Z","shell.execute_reply.started":"2024-07-20T06:40:44.033904Z","shell.execute_reply":"2024-07-20T06:41:44.142194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slice_count = files_df.groupby(['study_id', 'series_type']).size()\nplt.title('Number of slices for (study_id + series_description)')\nsns.histplot(slice_count, binwidth=5)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:44.145178Z","iopub.execute_input":"2024-07-20T06:41:44.145666Z","iopub.status.idle":"2024-07-20T06:41:44.519461Z","shell.execute_reply.started":"2024-07-20T06:41:44.145611Z","shell.execute_reply":"2024-07-20T06:41:44.518476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some series have very little amount of slices (< 10). For those cases we can duplicate some images the required number of times.","metadata":{}},{"cell_type":"code","source":"files_df.groupby('series_type').study_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:44.520970Z","iopub.execute_input":"2024-07-20T06:41:44.521710Z","iopub.status.idle":"2024-07-20T06:41:44.546048Z","shell.execute_reply.started":"2024-07-20T06:41:44.521673Z","shell.execute_reply":"2024-07-20T06:41:44.545018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"types_count = files_df.groupby('study_id').series_type.nunique()\npatients = types_count[types_count == 3].index\n\nfiles_df = files_df[files_df.study_id.isin(patients)]","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:44.548136Z","iopub.execute_input":"2024-07-20T06:41:44.548480Z","iopub.status.idle":"2024-07-20T06:41:44.574538Z","shell.execute_reply.started":"2024-07-20T06:41:44.548445Z","shell.execute_reply":"2024-07-20T06:41:44.573730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no Saggital T1/T2 series for some of the patients. For now let's drop those patients from the training process. As for inference let's fill those values with good quesses.","metadata":{}},{"cell_type":"code","source":"labels.melt(id_vars='study_id').groupby('variable').value.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:44.575681Z","iopub.execute_input":"2024-07-20T06:41:44.575976Z","iopub.status.idle":"2024-07-20T06:41:44.610865Z","shell.execute_reply.started":"2024-07-20T06:41:44.575952Z","shell.execute_reply":"2024-07-20T06:41:44.609908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting series fixed number of slices","metadata":{}},{"cell_type":"code","source":"def get_and_save_volume(study_id):\n    \n    study_filter = files_df.study_id == study_id\n    \n    volume = []\n    \n    for series_type in series_types:\n        \n        type_filter = files_df.series_type == series_type\n        series_df = files_df[study_filter & type_filter]\n        \n        step = (series_df.shape[0]- 1) / (Config.n_slices_per_series - 1)\n        \n        for i in range(Config.n_slices_per_series):\n            \n            idx = floor(i * step)\n            img = read_dcm_image('train_images', *series_df.iloc[idx, :3])\n            volume.append(img)\n    \n    volume = np.stack(volume)\n    np.save(f'volumes/{study_id}.npy', volume)","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:44:27.578766Z","iopub.execute_input":"2024-07-20T06:44:27.579141Z","iopub.status.idle":"2024-07-20T06:44:27.587255Z","shell.execute_reply.started":"2024-07-20T06:44:27.579106Z","shell.execute_reply":"2024-07-20T06:44:27.586197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_studies = files_df.study_id.unique()\ntasks = [delayed(get_and_save_volume)(study_id) for study_id in all_studies]","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:49.515187Z","iopub.execute_input":"2024-07-20T06:41:49.515665Z","iopub.status.idle":"2024-07-20T06:41:49.522414Z","shell.execute_reply.started":"2024-07-20T06:41:49.515634Z","shell.execute_reply":"2024-07-20T06:41:49.521295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Parallel(n_jobs=-1, verbose=10)(tasks)\npass","metadata":{"execution":{"iopub.status.busy":"2024-07-20T06:41:50.750341Z","iopub.execute_input":"2024-07-20T06:41:50.750719Z","iopub.status.idle":"2024-07-20T06:42:15.242999Z","shell.execute_reply.started":"2024-07-20T06:41:50.750690Z","shell.execute_reply":"2024-07-20T06:42:15.241864Z"},"trusted":true},"execution_count":null,"outputs":[]}]}