{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9581798,"sourceType":"datasetVersion","datasetId":5842667},{"sourceId":200023596,"sourceType":"kernelVersion"},{"sourceId":129660,"sourceType":"modelInstanceVersion","modelInstanceId":109250,"modelId":133565}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if not filename.endswith('.dcm'):\n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-08T23:53:05.688560Z","iopub.execute_input":"2024-10-08T23:53:05.688941Z","iopub.status.idle":"2024-10-08T23:53:33.784748Z","shell.execute_reply.started":"2024-10-08T23:53:05.688904Z","shell.execute_reply":"2024-10-08T23:53:33.783677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"This is my first time participating in a competition of this level on Kaggle, and I'm very happy to be here, learning from the amazing code shared by many experts. The code I submitted was based on Liam Nguyen's notebook: https://www.kaggle.com/code/namgalielei/lsdc-yolo-approach. I'm extremely grateful for his notebook, as it helped me understand the competition process!\"","metadata":{}},{"cell_type":"code","source":"\n\n# !pip install /kaggle/input/ultralytics/ultralytics-8.3.7-py3-none-any.whl ultralytics\n!pip install --no-index --find-links /kaggle/input/ultralytics ultralytics\n#!pip install ultralytics\n#!pip install ultralytics \n","metadata":{"execution":{"iopub.status.busy":"2024-10-09T00:23:05.139766Z","iopub.execute_input":"2024-10-09T00:23:05.140195Z","iopub.status.idle":"2024-10-09T00:23:18.094463Z","shell.execute_reply.started":"2024-10-09T00:23:05.140155Z","shell.execute_reply":"2024-10-09T00:23:18.093496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -rf /kaggle/working/wheel","metadata":{"execution":{"iopub.status.busy":"2024-10-09T00:05:50.356646Z","iopub.execute_input":"2024-10-09T00:05:50.357082Z","iopub.status.idle":"2024-10-09T00:05:51.406482Z","shell.execute_reply.started":"2024-10-09T00:05:50.357037Z","shell.execute_reply":"2024-10-09T00:05:51.405413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# !pip download ultralytics -d /kaggle/working/wheel --no-deps\n# !pip download ultralytics_thop -d /kaggle/working/wheel --no-deps\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-09T00:19:07.300983Z","iopub.execute_input":"2024-10-09T00:19:07.301755Z","iopub.status.idle":"2024-10-09T00:19:22.041575Z","shell.execute_reply.started":"2024-10-09T00:19:07.301675Z","shell.execute_reply":"2024-10-09T00:19:22.040409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\nfrom PIL import Image\nimport numpy as np\nfrom multiprocessing import Pool, cpu_count\n\nimport sklearn.metrics\nimport torch\nimport cv2\nimport numpy as np \nimport pandas as pd \nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.330295Z","iopub.execute_input":"2024-10-08T14:59:47.330662Z","iopub.status.idle":"2024-10-08T14:59:47.336624Z","shell.execute_reply.started":"2024-10-08T14:59:47.330625Z","shell.execute_reply":"2024-10-08T14:59:47.335683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EVAL = False # Change to True to compute the validation score\nIMG_DIR = '/images'\nFOLD = 0\nSAMPLE = False # True for quick debugging\nSEVERITIES = ['Normal/Mild', 'Moderate', 'Severe']\nLEVELS = ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']\n\nSCS_WEIGHTS = ['/kaggle/input/yolov8/pytorch/default/1/wight/scs_best.pt']\n\nSS_WEIGHTS = ['/kaggle/input/yolov8/pytorch/default/1/wight/ss_best.pt']\n#               ,\n#              '/kaggle/input/lsdc-yolo-ssv3/best.pt']\n\nNFN_WEIGHTS = ['/kaggle/input/yolov8/pytorch/default/1/wight/nfn_best.pt']\n#                ,\n#               '/kaggle/input/lsdc-yolo-nfnv3/best.pt']","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.339003Z","iopub.execute_input":"2024-10-08T14:59:47.339350Z","iopub.status.idle":"2024-10-08T14:59:47.345916Z","shell.execute_reply.started":"2024-10-08T14:59:47.339317Z","shell.execute_reply":"2024-10-08T14:59:47.345008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EVAL:\n    import sys\n    sys.path.append('/kaggle/input/lsdc-utils')\n    from metrics import score as lsdc_scoring","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.346889Z","iopub.execute_input":"2024-10-08T14:59:47.347225Z","iopub.status.idle":"2024-10-08T14:59:47.357477Z","shell.execute_reply.started":"2024-10-08T14:59:47.347193Z","shell.execute_reply":"2024-10-08T14:59:47.356708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ntrain_val_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.358486Z","iopub.execute_input":"2024-10-08T14:59:47.359321Z","iopub.status.idle":"2024-10-08T14:59:47.385918Z","shell.execute_reply.started":"2024-10-08T14:59:47.359274Z","shell.execute_reply":"2024-10-08T14:59:47.385197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nif EVAL:\n    train_xy = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\n    des = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\nelse:    \n    des = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.387014Z","iopub.execute_input":"2024-10-08T14:59:47.387368Z","iopub.status.idle":"2024-10-08T14:59:47.394297Z","shell.execute_reply.started":"2024-10-08T14:59:47.387333Z","shell.execute_reply":"2024-10-08T14:59:47.393134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_dcm(src_path):\n    dicom_data = pydicom.dcmread(src_path)\n    image = dicom_data.pixel_array\n    image = (image - image.min()) / (image.max() - image.min() +1e-6) * 255\n    return image\n\ndef convert_dcm_to_jpg(file_path):\n    try:\n        # Read the DICOM file\n        image_array = read_dcm(file_path)\n        \n        # Define the output path\n        relative_path = os.path.relpath(file_path, start=input_directory)\n        output_path = os.path.join(output_directory, relative_path)\n        output_path = output_path.replace('.dcm', '.jpg')\n                \n        # Create the output directory if it doesn't exist\n        os.makedirs(os.path.dirname(output_path), exist_ok=True)\n        \n        # Save the image as a JPEG file\n        cv2.imwrite(output_path, image_array)\n        \n        return output_path\n    except Exception as e:\n        print(f\"Error processing file {file_path}: {e}\")\n        return None\n\ndef process_files(dcm_files):\n    with Pool(cpu_count()) as pool:\n        # Wrap pool.map with tqdm to show the progress bar\n        list(tqdm(pool.imap(convert_dcm_to_jpg, dcm_files), total=len(dcm_files)))\n\ndef get_dcm_files(directory):\n    dcm_files = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            if file.endswith('.dcm'):\n                dcm_files.append(os.path.join(root, file))\n    return dcm_files    ","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.395590Z","iopub.execute_input":"2024-10-08T14:59:47.395977Z","iopub.status.idle":"2024-10-08T14:59:47.411628Z","shell.execute_reply.started":"2024-10-08T14:59:47.395929Z","shell.execute_reply":"2024-10-08T14:59:47.410596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace these with your input and output directories\nif not EVAL:\n    input_directory = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'\n\n    output_directory = IMG_DIR\n\n    # Get all .dcm files in the input directory\n    dcm_files = get_dcm_files(input_directory)\n\n    # Process the files using multiprocessing\n    process_files(dcm_files)\n\n    print(f\"Conversion completed. Images saved to {output_directory}\")\nelse:\n    if not os.path.exists(IMG_DIR):\n        print('Unziping data..')\n        !unzip -q -d / /kaggle/input/lsdc-get-all-images/images.zip\n        print('Done unziping data')","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:47.413004Z","iopub.execute_input":"2024-10-08T14:59:47.413360Z","iopub.status.idle":"2024-10-08T14:59:48.397404Z","shell.execute_reply.started":"2024-10-08T14:59:47.413311Z","shell.execute_reply":"2024-10-08T14:59:48.396310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EVAL:\n    fold_df = pd.read_csv('/kaggle/input/lsdc-fold-split/5folds.csv')\n    test_df = fold_df[fold_df.fold == FOLD]\n    \nelse:\n    test_df = os.listdir('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images')\n    test_df = pd.DataFrame(test_df, columns=['study_id'])\n    test_df['study_id'] = test_df['study_id'].astype(int)\n    \ntest_df = test_df.merge(des, on=['study_id'])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.401773Z","iopub.execute_input":"2024-10-08T14:59:48.402096Z","iopub.status.idle":"2024-10-08T14:59:48.413504Z","shell.execute_reply.started":"2024-10-08T14:59:48.402056Z","shell.execute_reply":"2024-10-08T14:59:48.412433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gen_label_map(CONDITIONS):\n    label2id = {}\n    id2label = {}\n    i = 0\n    for cond in CONDITIONS:\n        for level in LEVELS:\n            for severity in SEVERITIES:\n                cls_ = f\"{cond.lower().replace(' ', '_')}_{level}_{severity.lower()}\"\n                label2id[cls_] = i\n                id2label[i] = cls_\n                i+=1\n    return label2id, id2label\n                \nscs_label2id, scs_id2label = gen_label_map(['Spinal Canal Stenosis'])\nss_label2id, ss_id2label = gen_label_map(['Left Subarticular Stenosis', 'Right Subarticular Stenosis'])\nnfn_label2id, nfn_id2label = gen_label_map(['Left Neural Foraminal Narrowing', 'Right Neural Foraminal Narrowing'])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.414819Z","iopub.execute_input":"2024-10-08T14:59:48.415207Z","iopub.status.idle":"2024-10-08T14:59:48.443304Z","shell.execute_reply.started":"2024-10-08T14:59:48.415146Z","shell.execute_reply":"2024-10-08T14:59:48.442504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load YOLO Model\nscs_models = []\nfor weight in SCS_WEIGHTS:\n    scs_models.append(YOLO(weight))\n    \nss_models = []\nfor weight in SS_WEIGHTS:\n    ss_models.append(YOLO(weight))\n    \nnfn_models = []\nfor weight in NFN_WEIGHTS:\n    nfn_models.append(YOLO(weight))","metadata":{"execution":{"iopub.status.busy":"2024-10-08T15:07:42.964369Z","iopub.execute_input":"2024-10-08T15:07:42.964890Z","iopub.status.idle":"2024-10-08T15:07:43.227218Z","shell.execute_reply.started":"2024-10-08T15:07:42.964840Z","shell.execute_reply":"2024-10-08T15:07:43.226139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nall_label_set = train_val_df.iloc[0, 1:].index.tolist()\nscs_label_set = all_label_set[:5]\nnfn_label_set = all_label_set[5:15]\nss_label_set = all_label_set[15:]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.698053Z","iopub.execute_input":"2024-10-08T14:59:48.698471Z","iopub.status.idle":"2024-10-08T14:59:48.704049Z","shell.execute_reply.started":"2024-10-08T14:59:48.698426Z","shell.execute_reply":"2024-10-08T14:59:48.702981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nsettings = [\n    ( 'Sagittal T2/STIR', scs_models, scs_id2label, scs_label_set, 0.01),\n    ( 'Axial T2', ss_models, ss_id2label, ss_label_set, 0.01),\n    ( 'Sagittal T1', nfn_models, nfn_id2label, nfn_label_set, 0.1)\n]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.705695Z","iopub.execute_input":"2024-10-08T14:59:48.706067Z","iopub.status.idle":"2024-10-08T14:59:48.713749Z","shell.execute_reply.started":"2024-10-08T14:59:48.706020Z","shell.execute_reply":"2024-10-08T14:59:48.712832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.714985Z","iopub.execute_input":"2024-10-08T14:59:48.715338Z","iopub.status.idle":"2024-10-08T14:59:48.724713Z","shell.execute_reply.started":"2024-10-08T14:59:48.715296Z","shell.execute_reply":"2024-10-08T14:59:48.723741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_rows = []\n\nfor modality, models, id2label, label_set, thresh in settings:\n    mod_df = test_df[test_df.series_description == modality]\n    \n    if SAMPLE:\n        mod_df = mod_df.sample(20, random_state=610)\n    \n    # for each study, at each level and condition, get the maximum probability score\n    for study_id, group in tqdm(mod_df.groupby('study_id')):\n        predictions = defaultdict(list)\n        for i, row in group.iterrows():\n            # predict on all images from all the series\n            series_dir = os.path.join(IMG_DIR, str(row['study_id']), str(row['series_id']))\n            for model in models:\n                results = model(series_dir, conf=thresh, verbose=False)\n                for res in results:\n                    for pred_class, conf in zip(res.boxes.cls, res.boxes.conf):\n                        pred_class = pred_class.item()\n                        conf = conf.item()\n                        _class = id2label[pred_class]\n                        predictions[_class].append(conf)\n        \n        # aggregate the result on images to obtain study-level prediction\n        for condition in label_set:\n            res_dict = {'row_id': f'{study_id}_{condition}' }\n\n            score_vec = []\n            for severity in SEVERITIES:\n                severity = severity.lower()\n                key = f'{condition}_{severity}'\n                if len(predictions[key]) > 0:\n                    score = np.max(predictions[key])\n                else:\n                    score = thresh\n                score_vec.append(score)\n                \n            # normalize score to sum to 1\n            score_vec = torch.tensor(score_vec)\n            score_vec = score_vec / score_vec.sum()\n\n            for idx, severity in enumerate(SEVERITIES):\n                res_dict[severity.replace('/', '_').lower()] = score_vec[idx].item()\n\n            pred_rows.append(res_dict)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:48.726303Z","iopub.execute_input":"2024-10-08T14:59:48.726664Z","iopub.status.idle":"2024-10-08T14:59:50.457041Z","shell.execute_reply.started":"2024-10-08T14:59:48.726622Z","shell.execute_reply":"2024-10-08T14:59:50.456105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\npred_df = pd.DataFrame(pred_rows)\npred_df\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:50.458121Z","iopub.execute_input":"2024-10-08T14:59:50.458423Z","iopub.status.idle":"2024-10-08T14:59:50.474488Z","shell.execute_reply.started":"2024-10-08T14:59:50.458391Z","shell.execute_reply":"2024-10-08T14:59:50.473499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\npred_df.to_csv('submission.csv', index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:50.475798Z","iopub.execute_input":"2024-10-08T14:59:50.476116Z","iopub.status.idle":"2024-10-08T14:59:50.488162Z","shell.execute_reply.started":"2024-10-08T14:59:50.476071Z","shell.execute_reply":"2024-10-08T14:59:50.487287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_weight(row):\n    if row['normal_mild'] == 1:\n        return 1\n    if row['moderate'] == 1:\n        return 2\n    if row['severe'] == 1:\n        return 4\n    raise ValueError('No such value')\n    \ndef get_class(row):\n    return np.argmax([row['normal_mild'], row['moderate'], row['severe']])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:50.489312Z","iopub.execute_input":"2024-10-08T14:59:50.489606Z","iopub.status.idle":"2024-10-08T14:59:50.496128Z","shell.execute_reply.started":"2024-10-08T14:59:50.489574Z","shell.execute_reply":"2024-10-08T14:59:50.495425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EVAL:\n    gt_df = train_val_df.dropna().melt(id_vars=['study_id'], value_vars=all_label_set)\n    gt_df['row_id'] = gt_df['study_id'].astype(str) + '_' + gt_df['variable']\n    gt_df= gt_df[['row_id', 'value']]\n    gt_df = pd.get_dummies(gt_df, columns=['value'], dtype=int)\n    gt_df.columns = ['row_id', 'moderate', 'normal_mild', 'severe']\n    gt_df = gt_df[['row_id', 'normal_mild', 'moderate', 'severe']]\n    gt_df['sample_weight'] = gt_df.apply(sample_weight, axis=1)\n\n    gt_df1 = gt_df.merge(pred_df['row_id'], how='inner', on='row_id').sort_values('row_id').reset_index(drop=True)\n    pred_df1 = pred_df.merge(gt_df1['row_id'], how='inner', on='row_id').sort_values('row_id').reset_index(drop=True)\n    gt_df1['pred_cls'] = gt_df1.apply(get_class, axis=1)\n    pred_df1['pred_cls'] = pred_df1.apply(get_class, axis=1)\n\n    gt_df1[(gt_df1['pred_cls'] != pred_df1['pred_cls'])]\n    pred_df1[(gt_df1['pred_cls'] != pred_df1['pred_cls'])]\n    print('Label count:\\n', gt_df1['pred_cls'].value_counts(normalize=True))\n    print('Prediction accuracy:', (gt_df1['pred_cls'] == pred_df1['pred_cls']).mean())\n    print()\n\n    target_levels = ['normal_mild', 'moderate', 'severe']\n    loss = lsdc_scoring(gt_df1.drop(['pred_cls'], axis=1), pred_df1.drop(['pred_cls'], axis=1), row_id_column_name='row_id', any_severe_scalar=1)\n    print('Total weighted log loss:', loss)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:50.497301Z","iopub.execute_input":"2024-10-08T14:59:50.497790Z","iopub.status.idle":"2024-10-08T14:59:50.510419Z","shell.execute_reply.started":"2024-10-08T14:59:50.497745Z","shell.execute_reply":"2024-10-08T14:59:50.509526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n!ls\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T14:59:50.513519Z","iopub.execute_input":"2024-10-08T14:59:50.513878Z","iopub.status.idle":"2024-10-08T14:59:51.531480Z","shell.execute_reply.started":"2024-10-08T14:59:50.513831Z","shell.execute_reply":"2024-10-08T14:59:51.530423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}