{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":182785617,"sourceType":"kernelVersion"},{"sourceId":185212934,"sourceType":"kernelVersion"},{"sourceId":6124,"sourceType":"modelInstanceVersion","modelInstanceId":4599},{"sourceId":6125,"sourceType":"modelInstanceVersion","modelInstanceId":4596},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\n\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"  # @param [\"tensorflow\", \"jax\", \"torch\"]\n\nfrom tensorflow import data as tf_data\nimport tensorflow_datasets as tfds\nimport keras\nimport keras_cv\nimport numpy as np\nfrom keras_cv import bounding_box\nimport os\nfrom keras_cv import visualization\nimport tqdm\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-05T16:08:37.196528Z","iopub.execute_input":"2024-07-05T16:08:37.197011Z","iopub.status.idle":"2024-07-05T16:09:02.104781Z","shell.execute_reply.started":"2024-07-05T16:08:37.196976Z","shell.execute_reply":"2024-07-05T16:09:02.103429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\nTRAIN_DIR = BASE_DIR+'train_images/'\nTEST_DIR = BASE_DIR+'test_images/'\n\nPRETRAINED = 'efficientnetv2_b2_imagenet'\nBATCH_SIZE = 32\n\nIMG_SIZE = [320,320]","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.107637Z","iopub.execute_input":"2024-07-05T16:09:02.108533Z","iopub.status.idle":"2024-07-05T16:09:02.115311Z","shell.execute_reply.started":"2024-07-05T16:09:02.108483Z","shell.execute_reply":"2024-07-05T16:09:02.113696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies = os.listdir(TEST_DIR)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.116915Z","iopub.execute_input":"2024-07-05T16:09:02.117692Z","iopub.status.idle":"2024-07-05T16:09:02.133164Z","shell.execute_reply.started":"2024-07-05T16:09:02.117647Z","shell.execute_reply":"2024-07-05T16:09:02.131725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(BASE_DIR+'train.csv')\nlabels.study_id = labels.study_id.astype(str)\n\nconditions = np.unique(labels.columns[1:])\nclasses = []\nfor c in conditions:\n    classes.append(c+'_normal')\n    classes.append(c+'_moderate')\n    classes.append(c+'_severe')\nclasses_map = {classes[i]:i for i in range(len(classes))}\nclass_mapping = {i:classes[i] for i in range(len(classes))}\nN_CLASSES = len(class_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.136408Z","iopub.execute_input":"2024-07-05T16:09:02.13685Z","iopub.status.idle":"2024-07-05T16:09:02.185388Z","shell.execute_reply.started":"2024-07-05T16:09:02.136814Z","shell.execute_reply":"2024-07-05T16:09:02.184143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(studies):\n    image_paths = []\n    study_ids = []\n    for study_id in studies:\n        study_dir = TEST_DIR+study_id+'/'\n        for series_id in os.listdir(study_dir):\n            series_dir = study_dir+series_id+'/'\n            for z in os.listdir(series_dir):\n                path = series_dir+z\n                study_ids.append(study_id)\n                image_paths.append(path)\n    \n    return tf.data.Dataset.from_tensor_slices((np.array(image_paths), np.array(study_ids)))","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.186889Z","iopub.execute_input":"2024-07-05T16:09:02.187248Z","iopub.status.idle":"2024-07-05T16:09:02.194537Z","shell.execute_reply.started":"2024-07-05T16:09:02.187218Z","shell.execute_reply":"2024-07-05T16:09:02.193005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = prepare_data(studies)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.196182Z","iopub.execute_input":"2024-07-05T16:09:02.196572Z","iopub.status.idle":"2024-07-05T16:09:02.27971Z","shell.execute_reply.started":"2024-07-05T16:09:02.196537Z","shell.execute_reply":"2024-07-05T16:09:02.278462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x,y in test_data:\n    print(f\"{x}: {y}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:12:00.875635Z","iopub.execute_input":"2024-07-05T16:12:00.876907Z","iopub.status.idle":"2024-07-05T16:12:00.945992Z","shell.execute_reply.started":"2024-07-05T16:12:00.876853Z","shell.execute_reply":"2024-07-05T16:12:00.944825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    raw_image = tf.io.read_file(image_path)\n    sp = tf.strings.split(tf.gather(tf.strings.split(image_path, 'images/'), 1), '/')\n    N = tf.size(sp)\n    LEN = tf.strings.length(tf.gather(sp, 0))+tf.strings.length(tf.gather(sp, 2))\n    \n    # Add missing file metadata to avoid warnnigs flooding\n    if   LEN==12: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x92\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==13: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x92\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==14: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x94\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==15: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x94\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==16: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x96\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==17: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x96\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==18: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x98\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    \n    img = tfio.image.decode_dicom_image(raw_image, scale='auto', dtype=tf.float32)\n    m, M=tf.math.reduce_min(img), tf.math.reduce_max(img)\n    img = (tf.image.grayscale_to_rgb(img)-m)/(M-m)\n    img = tf.image.resize(img, IMG_SIZE)[0]\n    return img\n\ndef load_dataset(image_path, study_id):\n    image = load_image(image_path)\n    return {\"images\": tf.cast(image, tf.float32), \"study_id\":study_id}","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.281766Z","iopub.execute_input":"2024-07-05T16:09:02.282224Z","iopub.status.idle":"2024-07-05T16:09:02.29435Z","shell.execute_reply.started":"2024-07-05T16:09:02.282183Z","shell.execute_reply":"2024-07-05T16:09:02.29293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = test_data.map(load_dataset, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.ragged_batch(BATCH_SIZE, drop_remainder=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:02.295881Z","iopub.execute_input":"2024-07-05T16:09:02.296257Z","iopub.status.idle":"2024-07-05T16:09:03.595278Z","shell.execute_reply.started":"2024-07-05T16:09:02.296225Z","shell.execute_reply":"2024-07-05T16:09:03.593678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dict_to_tuple(inputs):\n    return inputs[\"images\"], inputs[\"study_id\"]\n\ntest_ds = test_ds.map(dict_to_tuple, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:03.596671Z","iopub.execute_input":"2024-07-05T16:09:03.597071Z","iopub.status.idle":"2024-07-05T16:09:03.649805Z","shell.execute_reply.started":"2024-07-05T16:09:03.597039Z","shell.execute_reply":"2024-07-05T16:09:03.64847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(PRETRAINED)\nmodel = keras.Sequential(\n    [\n        keras.layers.Input(shape=(None, None, 3)),\n        backbone,\n        keras.layers.GlobalMaxPooling2D(),\n        keras.layers.Dense(256, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(512, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(1024, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(N_CLASSES, activation=\"sigmoid\"),\n    ]\n)\nmodel.load_weights(\"/kaggle/input/rsna-keras-training-starter/best_model.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:24:18.026713Z","iopub.execute_input":"2024-07-05T16:24:18.027165Z","iopub.status.idle":"2024-07-05T16:24:27.615961Z","shell.execute_reply.started":"2024-07-05T16:24:18.027137Z","shell.execute_reply":"2024-07-05T16:24:27.613941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Pred = {}\npreds = []\nstudy_ids = []\nfor img, study_id in test_ds.as_numpy_iterator():\n    preds.append(model.predict(img))\n    study_ids.append(study_id)\n    \npreds = np.concatenate(preds)\nstudy_ids = np.concatenate(study_ids)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:24:27.61739Z","iopub.status.idle":"2024-07-05T16:24:27.617842Z","shell.execute_reply.started":"2024-07-05T16:24:27.617632Z","shell.execute_reply":"2024-07-05T16:24:27.617654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for p, study_id in zip(preds, study_ids):\n    study_id = str(int(study_id))\n    if study_id in Pred: Pred[study_id].append(p)\n    else: Pred[study_id] = []","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:24:27.619092Z","iopub.status.idle":"2024-07-05T16:24:27.619518Z","shell.execute_reply.started":"2024-07-05T16:24:27.619315Z","shell.execute_reply":"2024-07-05T16:24:27.619332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['normal_mild', 'moderate', 'severe'])\n\nfor study_id in Pred:\n    pred = np.array(Pred[study_id]).max(axis=0)\n    for i in range(N_CLASSES):\n        condition = '_'.join(class_mapping[i].split('_')[:-1])\n        intensity = class_mapping[i].split('_')[-1].replace('normal', 'normal_mild')\n        row_id = study_id+'_'+condition\n        submission.loc[row_id, intensity] = pred[i]\n\nsubmission = submission.reset_index().rename(columns={'index':'row_id'})","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:24:27.621123Z","iopub.status.idle":"2024-07-05T16:24:27.621515Z","shell.execute_reply.started":"2024-07-05T16:24:27.621332Z","shell.execute_reply":"2024-07-05T16:24:27.621348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:24:27.623344Z","iopub.status.idle":"2024-07-05T16:24:27.623747Z","shell.execute_reply.started":"2024-07-05T16:24:27.623538Z","shell.execute_reply":"2024-07-05T16:24:27.623553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T16:09:13.303657Z","iopub.status.idle":"2024-07-05T16:09:13.30408Z","shell.execute_reply.started":"2024-07-05T16:09:13.30389Z","shell.execute_reply":"2024-07-05T16:09:13.303907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Don't forget to upvote if you think this notebook was useful  😁","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}