{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":182785617,"sourceType":"kernelVersion"},{"sourceId":183482671,"sourceType":"kernelVersion"},{"sourceId":6124,"sourceType":"modelInstanceVersion","modelInstanceId":4599},{"sourceId":6125,"sourceType":"modelInstanceVersion","modelInstanceId":4596},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"  # @param [\"tensorflow\", \"jax\", \"torch\"]\n\nfrom tensorflow import data as tf_data\nimport tensorflow_datasets as tfds\nimport keras\nimport keras_cv\nimport numpy as np\nfrom keras_cv import bounding_box\nimport os\nfrom keras_cv import visualization\nimport tqdm\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-14T17:30:27.819224Z","iopub.execute_input":"2024-06-14T17:30:27.820151Z","iopub.status.idle":"2024-06-14T17:30:50.479763Z","shell.execute_reply.started":"2024-06-14T17:30:27.820111Z","shell.execute_reply":"2024-06-14T17:30:50.478631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\nTRAIN_DIR = BASE_DIR+'train_images/'\nTEST_DIR = BASE_DIR+'test_images/'\n\nPRETRAINED = 'efficientnetv2_b2_imagenet'\nBATCH_SIZE = 32\n\nIMG_SIZE = [320,320]","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:33:25.070543Z","iopub.execute_input":"2024-06-14T17:33:25.070961Z","iopub.status.idle":"2024-06-14T17:33:25.076890Z","shell.execute_reply.started":"2024-06-14T17:33:25.070903Z","shell.execute_reply":"2024-06-14T17:33:25.075547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies = os.listdir(TEST_DIR)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:50.493462Z","iopub.execute_input":"2024-06-14T17:30:50.493815Z","iopub.status.idle":"2024-06-14T17:30:50.506614Z","shell.execute_reply.started":"2024-06-14T17:30:50.493784Z","shell.execute_reply":"2024-06-14T17:30:50.505547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(BASE_DIR+'train.csv')\nlabels.study_id = labels.study_id.astype(str)\n\nconditions = np.unique(labels.columns[1:])\nclasses = []\nfor c in conditions:\n    classes.append(c+'_normal')\n    classes.append(c+'_moderate')\n    classes.append(c+'_severe')\nclasses_map = {classes[i]:i for i in range(len(classes))}\nclass_mapping = {i:classes[i] for i in range(len(classes))}\nN_CLASSES = len(class_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:50.507948Z","iopub.execute_input":"2024-06-14T17:30:50.508312Z","iopub.status.idle":"2024-06-14T17:30:50.551020Z","shell.execute_reply.started":"2024-06-14T17:30:50.508274Z","shell.execute_reply":"2024-06-14T17:30:50.550055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(studies):\n    image_paths = []\n    study_ids = []\n    for study_id in studies:\n        study_dir = TEST_DIR+study_id+'/'\n        for series_id in os.listdir(study_dir):\n            series_dir = study_dir+series_id+'/'\n            for z in os.listdir(series_dir):\n                path = series_dir+z\n                study_ids.append(study_id)\n                image_paths.append(path)\n    \n    return tf.data.Dataset.from_tensor_slices((np.array(image_paths), np.array(study_ids)))","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:50.552349Z","iopub.execute_input":"2024-06-14T17:30:50.552680Z","iopub.status.idle":"2024-06-14T17:30:50.559600Z","shell.execute_reply.started":"2024-06-14T17:30:50.552652Z","shell.execute_reply":"2024-06-14T17:30:50.558403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = prepare_data(studies)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:50.731782Z","iopub.execute_input":"2024-06-14T17:30:50.732721Z","iopub.status.idle":"2024-06-14T17:30:50.795225Z","shell.execute_reply.started":"2024-06-14T17:30:50.732685Z","shell.execute_reply":"2024-06-14T17:30:50.794238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    raw_image = tf.io.read_file(image_path)\n    sp = tf.strings.split(tf.gather(tf.strings.split(image_path, 'images/'), 1), '/')\n    N = tf.size(sp)\n    LEN = tf.strings.length(tf.gather(sp, 0))+tf.strings.length(tf.gather(sp, 2))\n    \n    # Add missing file metadata to avoid warnnigs flooding\n    if   LEN==12: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x92\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==13: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x92\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==14: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x94\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==15: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x94\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==16: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x96\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==17: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x96\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    elif LEN==18: raw_image = tf.strings.regex_replace(raw_image, pattern=b'DICM\\x02\\x00\\x01\\x00', rewrite=b'DICM\\x02\\x00\\x00\\x00UL\\x04\\x00\\x98\\x00\\x00\\x00\\x02\\x00\\x01\\x00')\n    \n    img = tfio.image.decode_dicom_image(raw_image, scale='auto', dtype=tf.float32)\n    m, M=tf.math.reduce_min(img), tf.math.reduce_max(img)\n    img = (tf.image.grayscale_to_rgb(img)-m)/(M-m)\n    img = tf.image.resize(img, IMG_SIZE)[0]\n    return img\n\ndef load_dataset(image_path, study_id):\n    image = load_image(image_path)\n    return {\"images\": tf.cast(image, tf.float32), \"study_id\":study_id}","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:51.233853Z","iopub.execute_input":"2024-06-14T17:30:51.234867Z","iopub.status.idle":"2024-06-14T17:30:51.246020Z","shell.execute_reply.started":"2024-06-14T17:30:51.234825Z","shell.execute_reply":"2024-06-14T17:30:51.244708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = test_data.map(load_dataset, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.ragged_batch(BATCH_SIZE, drop_remainder=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:52.637100Z","iopub.execute_input":"2024-06-14T17:30:52.637497Z","iopub.status.idle":"2024-06-14T17:30:53.883305Z","shell.execute_reply.started":"2024-06-14T17:30:52.637466Z","shell.execute_reply":"2024-06-14T17:30:53.882189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dict_to_tuple(inputs):\n    return inputs[\"images\"], inputs[\"study_id\"]\n\ntest_ds = test_ds.map(dict_to_tuple, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:30:53.884969Z","iopub.execute_input":"2024-06-14T17:30:53.885320Z","iopub.status.idle":"2024-06-14T17:30:53.933790Z","shell.execute_reply.started":"2024-06-14T17:30:53.885291Z","shell.execute_reply":"2024-06-14T17:30:53.932629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(PRETRAINED)\nmodel = keras.Sequential(\n    [\n        keras.layers.Input(shape=(None, None, 3)),\n        backbone,\n        keras.layers.GlobalMaxPooling2D(),\n        keras.layers.Dense(256, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(512, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(1024, activation='leaky_relu'),keras.layers.Dropout(rate=0.6),\n        keras.layers.Dense(N_CLASSES, activation=\"sigmoid\"),\n    ]\n)\nmodel.load_weights(\"/kaggle/input/rsna-keras-training-starter-tpu/best_model.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:33:55.387343Z","iopub.execute_input":"2024-06-14T17:33:55.387746Z","iopub.status.idle":"2024-06-14T17:34:04.862184Z","shell.execute_reply.started":"2024-06-14T17:33:55.387713Z","shell.execute_reply":"2024-06-14T17:34:04.860998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Pred = {}\npreds = []\nstudy_ids = []\nfor img, study_id in test_ds.as_numpy_iterator():\n    preds.append(model.predict(img))\n    study_ids.append(study_id)\n    \npreds = np.concatenate(preds)\nstudy_ids = np.concatenate(study_ids)","metadata":{"execution":{"iopub.status.busy":"2024-06-14T17:34:04.864087Z","iopub.execute_input":"2024-06-14T17:34:04.864432Z","iopub.status.idle":"2024-06-14T17:34:25.516132Z","shell.execute_reply.started":"2024-06-14T17:34:04.864403Z","shell.execute_reply":"2024-06-14T17:34:25.514986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for p, study_id in zip(preds, study_ids):\n    study_id = str(int(study_id))\n    if study_id in Pred: Pred[study_id].append(p)\n    else: Pred[study_id] = []","metadata":{"execution":{"iopub.status.busy":"2024-06-11T15:17:25.379491Z","iopub.execute_input":"2024-06-11T15:17:25.379853Z","iopub.status.idle":"2024-06-11T15:17:25.386188Z","shell.execute_reply.started":"2024-06-11T15:17:25.379824Z","shell.execute_reply":"2024-06-11T15:17:25.385017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['normal_mild', 'moderate', 'severe'])\n\nfor study_id in Pred:\n    pred = np.array(Pred[study_id]).max(axis=0)\n    for i in range(N_CLASSES):\n        condition = '_'.join(class_mapping[i].split('_')[:-1])\n        intensity = class_mapping[i].split('_')[-1].replace('normal', 'normal_mild')\n        row_id = study_id+'_'+condition\n        submission.loc[row_id, intensity] = pred[i]\n\nsubmission = submission.reset_index().rename(columns={'index':'row_id'})","metadata":{"execution":{"iopub.status.busy":"2024-06-11T15:18:02.862534Z","iopub.execute_input":"2024-06-11T15:18:02.863343Z","iopub.status.idle":"2024-06-11T15:18:02.890429Z","shell.execute_reply.started":"2024-06-11T15:18:02.863303Z","shell.execute_reply":"2024-06-11T15:18:02.889254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-06-11T15:18:05.079356Z","iopub.execute_input":"2024-06-11T15:18:05.080281Z","iopub.status.idle":"2024-06-11T15:18:05.094412Z","shell.execute_reply.started":"2024-06-11T15:18:05.080247Z","shell.execute_reply":"2024-06-11T15:18:05.093172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T14:06:31.568261Z","iopub.execute_input":"2024-06-10T14:06:31.568743Z","iopub.status.idle":"2024-06-10T14:06:31.585948Z","shell.execute_reply.started":"2024-06-10T14:06:31.568702Z","shell.execute_reply":"2024-06-10T14:06:31.584657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Don't forget to upvote if you think this notebook was useful  😁","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}