{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":71549,"databundleVersionId":8561470},{"sourceType":"datasetVersion","sourceId":9396719,"datasetId":5686852,"databundleVersionId":9598613}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Reading dicom files requires pydicom and gcdm. \nLack of internet access for submissions (in main competition page) means using wheels is necessary, not just for pydicom & gcdm but all other packages.\n\nWheel file names for download on pypi page:\n1. pydicom-3.0.0-py3-none-any.whl (https://pypi.org/project/pydicom/#files)\n2. python_gdcm-3.0.24.1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl (https://pypi.org/project/python-gdcm/#files)\n3. numpy-2.1.1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n4. pylibjpeg-2.0.1-py3-none-any.whl\n5. pylibjpeg_libjpeg-2.2.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n6. pylibjpeg_openjpeg-2.3.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n\n\nThen input these files in Kaggle.\n\nAs we proceed with further work and models, we can then add wheel files from packages here.","metadata":{}},{"cell_type":"markdown","source":"big aim: \nfor each row ID in the test set, you must predict a probability for each of the different severity levels","metadata":{}},{"cell_type":"code","source":"! pip install pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:26:46.00395Z","iopub.execute_input":"2024-09-15T08:26:46.004349Z","iopub.status.idle":"2024-09-15T08:27:02.668009Z","shell.execute_reply.started":"2024-09-15T08:26:46.004309Z","shell.execute_reply":"2024-09-15T08:27:02.66677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install numpy==1.26.4","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:27:02.670801Z","iopub.execute_input":"2024-09-15T08:27:02.671693Z","iopub.status.idle":"2024-09-15T08:27:17.507849Z","shell.execute_reply.started":"2024-09-15T08:27:02.671636Z","shell.execute_reply":"2024-09-15T08:27:17.506534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install pylibjpeg_libjpeg","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:27:17.509374Z","iopub.execute_input":"2024-09-15T08:27:17.509752Z","iopub.status.idle":"2024-09-15T08:27:33.447138Z","shell.execute_reply.started":"2024-09-15T08:27:17.509709Z","shell.execute_reply":"2024-09-15T08:27:33.445916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install pylibjpeg_openjpeg","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:27:33.450045Z","iopub.execute_input":"2024-09-15T08:27:33.450428Z","iopub.status.idle":"2024-09-15T08:27:49.203733Z","shell.execute_reply.started":"2024-09-15T08:27:33.45039Z","shell.execute_reply":"2024-09-15T08:27:49.202647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install python_gdcm","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:27:49.205326Z","iopub.execute_input":"2024-09-15T08:27:49.205713Z","iopub.status.idle":"2024-09-15T08:27:58.587845Z","shell.execute_reply.started":"2024-09-15T08:27:49.205673Z","shell.execute_reply":"2024-09-15T08:27:58.586471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for using packages inside the rsna2024-packages dataset i made:\n# path is '/kaggle/input/rsna2024-packages' when dataset is inputted\n\n#! pip install /kaggle/input/rsna2024-packages/*.whl\n!pip install numpy --no-index --find-links=file:///kaggle/input/rsna2024-packages/","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:27:58.589462Z","iopub.execute_input":"2024-09-15T08:27:58.589861Z","iopub.status.idle":"2024-09-15T08:28:13.274967Z","shell.execute_reply.started":"2024-09-15T08:27:58.58982Z","shell.execute_reply":"2024-09-15T08:28:13.273648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"packages = [\n    \"pydicom==3.0.0\",\n    \"python_gdcm==3.0.24.1\",\n    \"numpy==2.1.1\",\n    \"pylibjpeg==2.0.1\",\n    \"pylibjpeg_libjpeg==2.2.0\",\n    \"pylibjpeg_openjpeg==2.3.0\",\n    \"pandas==2.2.2\"\n]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:13.276747Z","iopub.execute_input":"2024-09-15T08:28:13.277158Z","iopub.status.idle":"2024-09-15T08:28:13.282778Z","shell.execute_reply.started":"2024-09-15T08:28:13.27712Z","shell.execute_reply":"2024-09-15T08:28:13.281576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from Claude. for other packages to maybe install\nimport os\nimport subprocess\n\ndef download_packages(packages, base_dir):\n    for package in packages:\n        # Extract package name (without version) for the directory name\n        package_name = package.split('==')[0]\n        \n        # Create a directory for each package\n        package_dir = os.path.join(base_dir, package_name)\n        os.makedirs(package_dir, exist_ok=True)\n        \n        # Construct and execute the pip download command\n        cmd = f\"pip download {package} -d {package_dir}\"\n        print(f\"Downloading {package}...\")\n        \n        try:\n            subprocess.run(cmd, shell=True, check=True)\n            print(f\"Successfully downloaded {package}\")\n        except subprocess.CalledProcessError as e:\n            print(f\"Error downloading {package}: {e}\")\n\n# Base directory for downloads\nbase_dir = \"./downloads\"\n\n# Run the download function\ndownload_packages(packages, base_dir)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:13.284203Z","iopub.execute_input":"2024-09-15T08:28:13.284499Z","iopub.status.idle":"2024-09-15T08:28:27.423708Z","shell.execute_reply.started":"2024-09-15T08:28:13.284466Z","shell.execute_reply":"2024-09-15T08:28:27.422605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:27.425074Z","iopub.execute_input":"2024-09-15T08:28:27.425424Z","iopub.status.idle":"2024-09-15T08:28:27.850252Z","shell.execute_reply.started":"2024-09-15T08:28:27.425386Z","shell.execute_reply":"2024-09-15T08:28:27.849378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ntrain_df.sort_values('study_id',ascending=True).head(10)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:27.854289Z","iopub.execute_input":"2024-09-15T08:28:27.855269Z","iopub.status.idle":"2024-09-15T08:28:27.92714Z","shell.execute_reply.started":"2024-09-15T08:28:27.855228Z","shell.execute_reply":"2024-09-15T08:28:27.926151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df.study_id==10728036]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:27.928612Z","iopub.execute_input":"2024-09-15T08:28:27.928984Z","iopub.status.idle":"2024-09-15T08:28:27.951394Z","shell.execute_reply.started":"2024-09-15T08:28:27.928946Z","shell.execute_reply":"2024-09-15T08:28:27.950309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the train_images contains a folder for each participant (unique study_id), e.g. 10728036. \n\neach study_id folder then contains 3 subfolders. likely about the conditions: \n\nCONDITIONS = {\n    \"Sagittal T2/STIR\": [\"Spinal Canal Stenosis\"],\n    \"Axial T2\": [\"Left Subarticular Stenosis\", \"Right Subarticular Stenosis\"],\n    \"Sagittal T1\": [\"Left Neural Foraminal Narrowing\", \"Right Neural Foraminal Narrowing\"],\n} (taken from https://www.kaggle.com/code/vsahin/3d-vit-single-stage-classifier-inference?scriptVersionId=196245856)\n\nhowever, what do the numbers of these subfolders mean? take 10728036 and we get the numbers 142859125, 2073726394, 2399638375, 3491739931 (this is 4 numbers even!)\n","metadata":{}},{"cell_type":"code","source":"train_label_coordinates = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ntrain_label_coordinates.sort_values('study_id',ascending=True).head(10)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:27.95291Z","iopub.execute_input":"2024-09-15T08:28:27.953387Z","iopub.status.idle":"2024-09-15T08:28:28.087603Z","shell.execute_reply.started":"2024-09-15T08:28:27.953339Z","shell.execute_reply":"2024-09-15T08:28:28.086517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_label_coordinates[train_label_coordinates.study_id==10728036].series_id.unique())\ntrain_label_coordinates[train_label_coordinates.study_id==10728036].sort_values('series_id',ascending=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:28.089238Z","iopub.execute_input":"2024-09-15T08:28:28.089782Z","iopub.status.idle":"2024-09-15T08:28:28.113654Z","shell.execute_reply.started":"2024-09-15T08:28:28.089733Z","shell.execute_reply":"2024-09-15T08:28:28.112665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the numbers of the subfolders correspond to the series_id. that being said, within series_id 142859125 there are 36 .dcm files, however the instance_number are not for all 36. in fact, there are just 5. what's the explanation for this?","metadata":{}},{"cell_type":"code","source":"print(len(train_label_coordinates[(train_label_coordinates.study_id==10728036) & (train_label_coordinates.series_id==142859125)].instance_number))\npatient_id = '10728036'\nseries_id = '142859125'\nprint(os.listdir(f'/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/{patient_id}/{series_id}'))\nprint(len(os.listdir(f'/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/{patient_id}/{series_id}')))","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:28.115002Z","iopub.execute_input":"2024-09-15T08:28:28.115321Z","iopub.status.idle":"2024-09-15T08:28:28.13118Z","shell.execute_reply.started":"2024-09-15T08:28:28.115285Z","shell.execute_reply":"2024-09-15T08:28:28.130209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"question remains after all this: how does one load the images for proper PyTorch/CV framework usage with using the csvs as reference?","metadata":{}},{"cell_type":"code","source":"train_series_descriptions = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\ntrain_series_descriptions[(train_series_descriptions.study_id==10728036)]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:28.132241Z","iopub.execute_input":"2024-09-15T08:28:28.132648Z","iopub.status.idle":"2024-09-15T08:28:28.155954Z","shell.execute_reply.started":"2024-09-15T08:28:28.132612Z","shell.execute_reply":"2024-09-15T08:28:28.155052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"seems like all series_id have their series_description? can we check if there happen to be any missing?","metadata":{}},{"cell_type":"markdown","source":"in the meantime, let's try to load an image.","metadata":{}},{"cell_type":"code","source":"idx = 3\ntrain_image_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/'\nstudy_dir = str(train_label_coordinates.study_id.iloc[idx]) + '/'\nseries_dir = str(train_label_coordinates.series_id.iloc[idx]) + '/'\n\nfull_dir = os.path.join(train_image_dir,study_dir,series_dir,str(f\"1.dcm\"))\nprint(full_dir)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:28.157214Z","iopub.execute_input":"2024-09-15T08:28:28.157538Z","iopub.status.idle":"2024-09-15T08:28:28.16417Z","shell.execute_reply.started":"2024-09-15T08:28:28.157502Z","shell.execute_reply":"2024-09-15T08:28:28.163077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm_img = pydicom.dcmread(full_dir, force=True)\ndcm_img","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:28:28.165659Z","iopub.execute_input":"2024-09-15T08:28:28.16626Z","iopub.status.idle":"2024-09-15T08:28:28.235111Z","shell.execute_reply.started":"2024-09-15T08:28:28.166212Z","shell.execute_reply":"2024-09-15T08:28:28.233397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_series_descriptions = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\ntrain_series_descriptions[train_series_descriptions.study_id==10728036]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:32:51.512654Z","iopub.execute_input":"2024-09-15T08:32:51.513085Z","iopub.status.idle":"2024-09-15T08:32:51.534459Z","shell.execute_reply.started":"2024-09-15T08:32:51.513044Z","shell.execute_reply":"2024-09-15T08:32:51.533432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}