{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":8728463,"sourceType":"datasetVersion","datasetId":5238658},{"sourceId":8963992,"sourceType":"datasetVersion","datasetId":5395594},{"sourceId":184402550,"sourceType":"kernelVersion"}],"dockerImageVersionId":30733,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA2024 LSDC Submission Baseline\nIn the [previous notebook](https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-training-baseline), We trained the models.\n\nThis notebook will Let the model infer and make a submission.\n\n### My other Notebooks\n- [RSNA2024 LSDC Making Dataset](https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-making-dataset) \n- [RSNA2024 LSDC Training Baseline](https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-training-baseline) \n- [RSNA2024 LSDC Submission Baseline](https://www.kaggle.com/code/itsuki9180/rsna2024-lsdc-submission-baseline) <- you're reading now","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Import Libralies","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nfrom PIL import Image\nimport cv2\nimport math, random\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import KFold\n\nfrom collections import OrderedDict\n\nimport torch\nimport torch.nn.functional as F\nfrom torch import nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.optim import AdamW\n\nimport timm\nfrom timm.utils import ModelEmaV2\nfrom transformers import get_cosine_schedule_with_warmup\n\nimport albumentations as A\n\nfrom sklearn.model_selection import KFold\n\nimport re\nimport pydicom","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:00.962453Z","iopub.execute_input":"2024-07-16T05:06:00.963068Z","iopub.status.idle":"2024-07-16T05:06:00.970453Z","shell.execute_reply.started":"2024-07-16T05:06:00.963038Z","shell.execute_reply":"2024-07-16T05:06:00.969412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rd = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification'","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:01.878932Z","iopub.execute_input":"2024-07-16T05:06:01.879761Z","iopub.status.idle":"2024-07-16T05:06:01.883933Z","shell.execute_reply.started":"2024-07-16T05:06:01.879727Z","shell.execute_reply":"2024-07-16T05:06:01.882911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"OUTPUT_DIR = f'/kaggle/input/rsna2024-lsdc-training-baseline/rsna24-results'\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nN_WORKERS = os.cpu_count()\nUSE_AMP = True\nSEED = 8620\n\nIMG_SIZE = [512, 512]\nIN_CHANS = 30\nN_LABELS = 25\nN_CLASSES = 3 * N_LABELS\n\nN_FOLDS = 5\n\n# MODEL_NAME = \"tf_efficientnet_b3.ns_jft_in1k\"\n\nMODEL_NAME = \"densenet201\"\n\nBATCH_SIZE = 1","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:03.703006Z","iopub.execute_input":"2024-07-16T05:06:03.703408Z","iopub.status.idle":"2024-07-16T05:06:03.757565Z","shell.execute_reply.started":"2024-07-16T05:06:03.703377Z","shell.execute_reply":"2024-07-16T05:06:03.756477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rd = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification'","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:06.389239Z","iopub.execute_input":"2024-07-16T05:06:06.390135Z","iopub.status.idle":"2024-07-16T05:06:06.394393Z","shell.execute_reply.started":"2024-07-16T05:06:06.390076Z","shell.execute_reply":"2024-07-16T05:06:06.393344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda:0') if torch.cuda.is_available() else torch.device('cpu')\ndevice","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:07.832563Z","iopub.execute_input":"2024-07-16T05:06:07.833185Z","iopub.status.idle":"2024-07-16T05:06:07.840383Z","shell.execute_reply.started":"2024-07-16T05:06:07.833154Z","shell.execute_reply":"2024-07-16T05:06:07.839455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f'{rd}/test_series_descriptions.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:08.761845Z","iopub.execute_input":"2024-07-16T05:06:08.762259Z","iopub.status.idle":"2024-07-16T05:06:08.78903Z","shell.execute_reply.started":"2024-07-16T05:06:08.762227Z","shell.execute_reply":"2024-07-16T05:06:08.788082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_ids = list(df['study_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:10.708729Z","iopub.execute_input":"2024-07-16T05:06:10.709092Z","iopub.status.idle":"2024-07-16T05:06:10.7163Z","shell.execute_reply.started":"2024-07-16T05:06:10.709063Z","shell.execute_reply":"2024-07-16T05:06:10.715236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(f'{rd}/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:11.830432Z","iopub.execute_input":"2024-07-16T05:06:11.831238Z","iopub.status.idle":"2024-07-16T05:06:11.841216Z","shell.execute_reply.started":"2024-07-16T05:06:11.831205Z","shell.execute_reply":"2024-07-16T05:06:11.840283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LABELS = list(sample_sub.columns[1:])\nLABELS","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:12.218312Z","iopub.execute_input":"2024-07-16T05:06:12.218962Z","iopub.status.idle":"2024-07-16T05:06:12.225503Z","shell.execute_reply.started":"2024-07-16T05:06:12.218931Z","shell.execute_reply":"2024-07-16T05:06:12.224394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONDITIONS = [\n    'spinal_canal_stenosis', \n    'left_neural_foraminal_narrowing', \n    'right_neural_foraminal_narrowing',\n    'left_subarticular_stenosis',\n    'right_subarticular_stenosis'\n]\n\nLEVELS = [\n    'l1_l2',\n    'l2_l3',\n    'l3_l4',\n    'l4_l5',\n    'l5_s1',\n]","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:13.339088Z","iopub.execute_input":"2024-07-16T05:06:13.339906Z","iopub.status.idle":"2024-07-16T05:06:13.344652Z","shell.execute_reply.started":"2024-07-16T05:06:13.339877Z","shell.execute_reply":"2024-07-16T05:06:13.343576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def atoi(text):\n    return int(text) if text.isdigit() else text\n\ndef natural_keys(text):\n    return [ atoi(c) for c in re.split(r'(\\d+)', text) ]","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:14.358985Z","iopub.execute_input":"2024-07-16T05:06:14.359829Z","iopub.status.idle":"2024-07-16T05:06:14.364746Z","shell.execute_reply.started":"2024-07-16T05:06:14.359799Z","shell.execute_reply":"2024-07-16T05:06:14.36375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Dataset","metadata":{}},{"cell_type":"code","source":"class RSNA24TestDataset(Dataset):\n    def __init__(self, df, study_ids, phase='test', transform=None):\n        self.df = df\n        self.study_ids = study_ids\n        self.transform = transform\n        self.phase = phase\n    \n    def __len__(self):\n        return len(self.study_ids)\n    \n    def get_img_paths(self, study_id, series_desc):\n        pdf = self.df[self.df['study_id']==study_id]\n        pdf_ = pdf[pdf['series_description']==series_desc]\n        allimgs = []\n        for i, row in pdf_.iterrows():\n            pimgs = glob.glob(f'{rd}/test_images/{study_id}/{row[\"series_id\"]}/*.dcm')\n            pimgs = sorted(pimgs, key=natural_keys)\n            allimgs.extend(pimgs)\n            \n        return allimgs\n    \n    def read_dcm_ret_arr(self, src_path):\n        dicom_data = pydicom.dcmread(src_path)\n        image = dicom_data.pixel_array\n        image = (image - image.min()) / (image.max() - image.min() + 1e-6) * 255\n        img = cv2.resize(image, (IMG_SIZE[0], IMG_SIZE[1]),interpolation=cv2.INTER_CUBIC)\n        assert img.shape==(IMG_SIZE[0], IMG_SIZE[1])\n        return img\n\n    def __getitem__(self, idx):\n        x = np.zeros((IMG_SIZE[0], IMG_SIZE[1], IN_CHANS), dtype=np.uint8)\n        st_id = self.study_ids[idx]        \n        \n        # Sagittal T1\n        allimgs_st1 = self.get_img_paths(st_id, 'Sagittal T1')\n        if len(allimgs_st1)==0:\n            print(st_id, ': Sagittal T1, has no images')\n        \n        else:\n            step = len(allimgs_st1) / 10.0\n            st = len(allimgs_st1)/2.0 - 4.0*step\n            end = len(allimgs_st1)+0.0001\n            for j, i in enumerate(np.arange(st, end, step)):\n                try:\n                    ind2 = max(0, int((i-0.5001).round()))\n                    img = self.read_dcm_ret_arr(allimgs_st1[ind2])\n                    x[..., j] = img.astype(np.uint8)\n                except:\n                    print(f'failed to load on {st_id}, Sagittal T1')\n                    pass\n            \n        # Sagittal T2/STIR\n        allimgs_st2 = self.get_img_paths(st_id, 'Sagittal T2/STIR')\n        if len(allimgs_st2)==0:\n            print(st_id, ': Sagittal T2/STIR, has no images')\n            \n        else:\n            step = len(allimgs_st2) / 10.0\n            st = len(allimgs_st2)/2.0 - 4.0*step\n            end = len(allimgs_st2)+0.0001\n            for j, i in enumerate(np.arange(st, end, step)):\n                try:\n                    ind2 = max(0, int((i-0.5001).round()))\n                    img = self.read_dcm_ret_arr(allimgs_st2[ind2])\n                    x[..., j+10] = img.astype(np.uint8)\n                except:\n                    print(f'failed to load on {st_id}, Sagittal T2/STIR')\n                    pass\n            \n        # Axial T2\n        allimgs_at2 = self.get_img_paths(st_id, 'Axial T2')\n        if len(allimgs_at2)==0:\n            print(st_id, ': Axial T2, has no images')\n            \n        else:\n            step = len(allimgs_at2) / 10.0\n            st = len(allimgs_at2)/2.0 - 4.0*step\n            end = len(allimgs_at2)+0.0001\n\n            for j, i in enumerate(np.arange(st, end, step)):\n                try:\n                    ind2 = max(0, int((i-0.5001).round()))\n                    img = self.read_dcm_ret_arr(allimgs_at2[ind2])\n                    x[..., j+20] = img.astype(np.uint8)\n                except:\n                    print(f'failed to load on {st_id}, Axial T2')\n                    pass  \n            \n            \n        if self.transform is not None:\n            x = self.transform(image=x)['image']\n\n        x = x.transpose(2, 0, 1)\n                \n        return x, str(st_id)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:19.160127Z","iopub.execute_input":"2024-07-16T05:06:19.160517Z","iopub.status.idle":"2024-07-16T05:06:19.182695Z","shell.execute_reply.started":"2024-07-16T05:06:19.160488Z","shell.execute_reply":"2024-07-16T05:06:19.181486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transforms_test = A.Compose([\n    A.Resize(IMG_SIZE[0], IMG_SIZE[1]),\n    A.Normalize(mean=0.5, std=0.5)\n])","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:20.05408Z","iopub.execute_input":"2024-07-16T05:06:20.054505Z","iopub.status.idle":"2024-07-16T05:06:20.060186Z","shell.execute_reply.started":"2024-07-16T05:06:20.054473Z","shell.execute_reply":"2024-07-16T05:06:20.059077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = RSNA24TestDataset(df, study_ids, transform=transforms_test)\ntest_dl = DataLoader(\n    test_ds, \n    batch_size=1, \n    shuffle=False,\n    num_workers=N_WORKERS,\n    pin_memory=True,\n    drop_last=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:06:20.662778Z","iopub.execute_input":"2024-07-16T05:06:20.663162Z","iopub.status.idle":"2024-07-16T05:06:20.668502Z","shell.execute_reply.started":"2024-07-16T05:06:20.663129Z","shell.execute_reply":"2024-07-16T05:06:20.667623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Model","metadata":{}},{"cell_type":"code","source":"class ArcNet(nn.Module):\n    def __init__(self, feature_num, cls_num):\n        super(ArcNet, self).__init__()\n        self.w = nn.Parameter(torch.randn((feature_num, cls_num)), requires_grad=True)\n\n    def forward(self, x, s=30, m=0.3):\n        x_norm = nn.functional.normalize(x, dim=1)\n        w_norm = nn.functional.normalize(self.w, dim=0)\n        cosa = torch.matmul(x_norm, w_norm) / s\n        a = torch.acos(cosa)\n        arcsoftmax = torch.exp(\n            s * torch.cos(a + m)) / (torch.sum(torch.exp(s * cosa), dim=1, keepdim=True) - torch.exp(\n            s * cosa) + torch.exp(s * torch.cos(a + m)))\n\n        return torch.log(arcsoftmax)\n\nclass RSNA24Model(nn.Module):\n    def __init__(self, model_name, in_c=30, n_classes=75, pretrained=True, features_only=False):\n        super().__init__()\n        self.model = timm.create_model(\n                                    model_name,\n                                    pretrained=pretrained, \n                                    features_only=features_only,\n                                    in_chans=in_c,\n                                    num_classes=n_classes,\n                                    global_pool='avg'\n                                    )\n        in_features = self.model.get_classifier().in_features\n        self.model.classifier = nn.Sequential(\n            nn.Linear(in_features, in_features),\n            nn.RReLU(inplace=True),\n            nn.Dropout(0.5),\n            ArcNet(in_features, n_classes)\n#             nn.Linear(in_features,n_classes)\n        )\n    \n    def forward(self, x):\n        y = self.model(x)\n        return y","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:09:12.257112Z","iopub.execute_input":"2024-07-16T05:09:12.257811Z","iopub.status.idle":"2024-07-16T05:09:12.269839Z","shell.execute_reply.started":"2024-07-16T05:09:12.257774Z","shell.execute_reply":"2024-07-16T05:09:12.268715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Models","metadata":{}},{"cell_type":"code","source":"import glob\nmodels = []\n\nmodel = RSNA24Model(MODEL_NAME, IN_CHANS, N_CLASSES, pretrained=False)\nmodel.load_state_dict(torch.load('/kaggle/input/best-densenet-fold-0-arcface/best_wll_model_fold-0_arcface.pt'))\nmodel.eval()\nmodel.half()\nmodel.to(device)\nmodels.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:09:38.786436Z","iopub.execute_input":"2024-07-16T05:09:38.786857Z","iopub.status.idle":"2024-07-16T05:09:39.79136Z","shell.execute_reply.started":"2024-07-16T05:09:38.786806Z","shell.execute_reply":"2024-07-16T05:09:39.790522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = []","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:07:15.420198Z","iopub.execute_input":"2024-07-16T05:07:15.420811Z","iopub.status.idle":"2024-07-16T05:07:15.425035Z","shell.execute_reply.started":"2024-07-16T05:07:15.420779Z","shell.execute_reply":"2024-07-16T05:07:15.42409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nCKPT_PATHS = glob.glob('/kaggle/input/rsna2024-lsdc-training-baseline/rsna24-results/best_wll_model_fold-*.pt')\nCKPT_PATHS = sorted(CKPT_PATHS)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T20:02:30.161539Z","iopub.execute_input":"2024-06-19T20:02:30.161842Z","iopub.status.idle":"2024-06-19T20:02:30.174615Z","shell.execute_reply.started":"2024-06-19T20:02:30.16181Z","shell.execute_reply":"2024-06-19T20:02:30.173816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, cp in enumerate(CKPT_PATHS):\n    print(f'loading {cp}...')\n    model = RSNA24Model(MODEL_NAME, IN_CHANS, N_CLASSES, pretrained=False)\n    model.load_state_dict(torch.load(cp))\n    model.eval()\n    model.half()\n    model.to(device)\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T20:02:30.175646Z","iopub.execute_input":"2024-06-19T20:02:30.176918Z","iopub.status.idle":"2024-06-19T20:02:36.790225Z","shell.execute_reply.started":"2024-06-19T20:02:30.176894Z","shell.execute_reply":"2024-06-19T20:02:36.789374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference loop","metadata":{}},{"cell_type":"code","source":"autocast = torch.cuda.amp.autocast(enabled=USE_AMP, dtype=torch.half)\ny_preds = []\nrow_names = []\n\nwith tqdm(test_dl, leave=True) as pbar:\n    with torch.no_grad():\n        for idx, (x, si) in enumerate(pbar):\n            x = x.to(device)\n            pred_per_study = np.zeros((25, 3))\n            \n            for cond in CONDITIONS:\n                for level in LEVELS:\n                    row_names.append(si[0] + '_' + cond + '_' + level)\n            \n            with autocast:\n                for m in models:\n                    y = m(x)[0]\n                    for col in range(N_LABELS):\n                        pred = y[col*3:col*3+3]\n                        y_pred = pred.float().softmax(0).cpu().numpy()\n                        pred_per_study[col] += y_pred / len(models)\n                y_preds.append(pred_per_study)\n\ny_preds = np.concatenate(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:09:42.59465Z","iopub.execute_input":"2024-07-16T05:09:42.595015Z","iopub.status.idle":"2024-07-16T05:09:45.679761Z","shell.execute_reply.started":"2024-07-16T05:09:42.594987Z","shell.execute_reply":"2024-07-16T05:09:45.678621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Submission","metadata":{}},{"cell_type":"code","source":"sub = pd.DataFrame()\nsub['row_id'] = row_names\nsub[LABELS] = y_preds\nsub.head(25)","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:09:45.681726Z","iopub.execute_input":"2024-07-16T05:09:45.682074Z","iopub.status.idle":"2024-07-16T05:09:45.712077Z","shell.execute_reply.started":"2024-07-16T05:09:45.682045Z","shell.execute_reply":"2024-07-16T05:09:45.710946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)\npd.read_csv('submission.csv').head()","metadata":{"execution":{"iopub.status.busy":"2024-07-16T05:09:51.520986Z","iopub.execute_input":"2024-07-16T05:09:51.521677Z","iopub.status.idle":"2024-07-16T05:09:51.542119Z","shell.execute_reply.started":"2024-07-16T05:09:51.521646Z","shell.execute_reply":"2024-07-16T05:09:51.540776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion\nWe created the dataset, performed training, and inference in this notebook. \n\nThis competition is a bit complicated to handle the dataset, so there may be a better way.\n\nI think there are many other areas to improve in my notebook. I hope you can learn from my notebook and get a better score.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}