{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":123715,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":104132,"modelId":128334}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport pandas as pd \nimport pydicom\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:14.095047Z","iopub.execute_input":"2024-09-30T04:29:14.095744Z","iopub.status.idle":"2024-09-30T04:29:21.477669Z","shell.execute_reply.started":"2024-09-30T04:29:14.095684Z","shell.execute_reply":"2024-09-30T04:29:21.475482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the path to the dataset\n\ntrain_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\n\ntest_description = pd.read_csv(train_path + 'test_series_descriptions.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:37.985510Z","iopub.execute_input":"2024-09-30T04:29:37.986038Z","iopub.status.idle":"2024-09-30T04:29:37.995660Z","shell.execute_reply.started":"2024-09-30T04:29:37.985990Z","shell.execute_reply":"2024-09-30T04:29:37.994098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_description.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:38.751770Z","iopub.execute_input":"2024-09-30T04:29:38.752274Z","iopub.status.idle":"2024-09-30T04:29:38.786755Z","shell.execute_reply.started":"2024-09-30T04:29:38.752230Z","shell.execute_reply":"2024-09-30T04:29:38.785367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\n\n# Function to generate image paths based on directory structure\ndef generate_image_paths(df, data_dir):\n    image_paths = []\n    for study_id, series_id in zip(df['study_id'], df['series_id']):\n        study_dir = os.path.join(data_dir, str(study_id))\n        series_dir = os.path.join(study_dir, str(series_id))\n        images = os.listdir(series_dir)\n        image_paths.extend([os.path.join(series_dir, img) for img in images])\n    return image_paths\n\n\ntest_image_paths = generate_image_paths(test_description, f'{train_path}/test_images')\n\n# Function to visualize images using OpenCV\ndef visualize_image(image_path):\n    # Read the image using OpenCV\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    \n    # OpenCV reads images in BGR format, so no need to convert for grayscale\n    if img is None:\n        print(f\"Image not found at path: {image_path}\")\n        return\n    \n    # Display the image using Matplotlib for better color support\n    plt.imshow(img, cmap='rgb')\n    plt.title(f\"Image: {os.path.basename(image_path)}\")\n    plt.axis('off')  # Hide axis for better visualization\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:39.304585Z","iopub.execute_input":"2024-09-30T04:29:39.305078Z","iopub.status.idle":"2024-09-30T04:29:39.632463Z","shell.execute_reply.started":"2024-09-30T04:29:39.305033Z","shell.execute_reply":"2024-09-30T04:29:39.631317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## mapping conditions","metadata":{}},{"cell_type":"code","source":"condition_mapping = {\n    'Sagittal T1': {'left': 'left_neural_foraminal_narrowing', 'right': 'right_neural_foraminal_narrowing'},\n    'Axial T2': {'left': 'left_subarticular_stenosis', 'right': 'right_subarticular_stenosis'},\n    'Sagittal T2/STIR': 'spinal_canal_stenosis'\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:40.301546Z","iopub.execute_input":"2024-09-30T04:29:40.302023Z","iopub.status.idle":"2024-09-30T04:29:40.308128Z","shell.execute_reply.started":"2024-09-30T04:29:40.301979Z","shell.execute_reply":"2024-09-30T04:29:40.306864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images/'\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:40.703392Z","iopub.execute_input":"2024-09-30T04:29:40.703876Z","iopub.status.idle":"2024-09-30T04:29:40.709159Z","shell.execute_reply.started":"2024-09-30T04:29:40.703831Z","shell.execute_reply":"2024-09-30T04:29:40.708207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_paths(row):\n    series_path = os.path.join(base_path, str(row['study_id']), str(row['series_id']))\n    if os.path.exists(series_path):\n        return [os.path.join(series_path, f) for f in os.listdir(series_path) if os.path.isfile(os.path.join(series_path, f))]\n    return []\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:41.114796Z","iopub.execute_input":"2024-09-30T04:29:41.115277Z","iopub.status.idle":"2024-09-30T04:29:41.122589Z","shell.execute_reply.started":"2024-09-30T04:29:41.115232Z","shell.execute_reply":"2024-09-30T04:29:41.121265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expanded_rows = []\nfor index, row in test_description.iterrows():\n    image_paths = get_image_paths(row)\n    conditions = condition_mapping.get(row['series_description'], {})\n    if isinstance(conditions, str):  # Single condition\n        conditions = {'left': conditions, 'right': conditions}\n    for side, condition in conditions.items():\n        for image_path in image_paths:\n            expanded_rows.append({\n                'study_id': row['study_id'],\n                'series_id': row['series_id'],\n                'series_description': row['series_description'],\n                'image_path': image_path,\n                'condition': condition,\n                'row_id': f\"{row['study_id']}_{condition}\"\n            })\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:41.638749Z","iopub.execute_input":"2024-09-30T04:29:41.639243Z","iopub.status.idle":"2024-09-30T04:29:41.659030Z","shell.execute_reply.started":"2024-09-30T04:29:41.639190Z","shell.execute_reply":"2024-09-30T04:29:41.657734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(expanded_rows)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:42.033859Z","iopub.execute_input":"2024-09-30T04:29:42.034371Z","iopub.status.idle":"2024-09-30T04:29:42.041616Z","shell.execute_reply.started":"2024-09-30T04:29:42.034318Z","shell.execute_reply":"2024-09-30T04:29:42.040244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Levels for row_id\nlevels = ['l1_l2', 'l2_l3', 'l3_l4', 'l4_l5', 'l5_s1']\n\n# update row_id with levels\ndef update_row_id(row, levels):\n    level = levels[row.name % len(levels)]  \n    return f\"{row['study_id']}_{row['condition']}_{level}\"\n\n# Update row_id in expanded_test_desc to include levels\ntest_df['row_id'] = test_df.apply(lambda row: update_row_id(row, levels), axis=1)\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:42.678251Z","iopub.execute_input":"2024-09-30T04:29:42.678732Z","iopub.status.idle":"2024-09-30T04:29:42.707267Z","shell.execute_reply.started":"2024-09-30T04:29:42.678688Z","shell.execute_reply":"2024-09-30T04:29:42.705866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataset:\n    def __init__(self, dataframe, batch_size=16, image_size=(256, 256), normalize=False):\n        self.dataframe = dataframe\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.normalize = normalize\n\n    def load_image(self, image_path):\n        if image_path.lower().endswith('.dcm'):\n            dicom = pydicom.dcmread(image_path, force=True)\n            image = dicom.pixel_array\n        else:\n            image = cv2.imread(image_path, cv2.IMREAD_UNCHANGED)\n            if image is None:\n                raise FileNotFoundError(f\"Could not load image from {image_path}\")\n\n        # Convert image to uint8 if necessary\n        if image.dtype != np.uint8:\n            image = image.astype(np.uint8)\n\n        # If the image is grayscale, stack it to make 3 channels\n        if len(image.shape) == 2:\n            image = np.stack([image] * 3, axis=-1)\n\n        # Normalize the image if the flag is set\n        if self.normalize:\n            image = image / 255.0  # Normalization to [0, 1]\n\n        return image\n\n    def __getitem__(self, index):\n        # Get the starting index for this batch\n        start_index = index * self.batch_size\n        end_index = min((index + 1) * self.batch_size, len(self.dataframe))\n\n        images = []\n        row_ids = []\n\n        for i in range(start_index, end_index):\n            row = self.dataframe.iloc[i]\n            image_path = row['image_path']\n            row_id = row['row_id']\n\n            # Load and resize image\n            image = self.load_image(image_path)\n            image = cv2.resize(image, self.image_size)\n\n            images.append(image)\n            row_ids.append(row_id)\n\n        # Convert the list of images and row_ids to numpy arrays\n        images = np.array(images)\n        row_ids = np.array(row_ids)\n\n        return images, row_ids\n\n    def __len__(self):\n        # Number of batches per epoch\n        return int(np.ceil(len(self.dataframe) / self.batch_size))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:43.225488Z","iopub.execute_input":"2024-09-30T04:29:43.226737Z","iopub.status.idle":"2024-09-30T04:29:43.244289Z","shell.execute_reply.started":"2024-09-30T04:29:43.226683Z","shell.execute_reply":"2024-09-30T04:29:43.242784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = TestDataset(test_df,batch_size = 16, image_size=(256, 256), normalize=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:43.796684Z","iopub.execute_input":"2024-09-30T04:29:43.797146Z","iopub.status.idle":"2024-09-30T04:29:43.802918Z","shell.execute_reply.started":"2024-09-30T04:29:43.797103Z","shell.execute_reply":"2024-09-30T04:29:43.801619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the saved model (make sure this is the correct path to your model)\nfrom tensorflow import keras\nmodel = keras.models.load_model(\"/kaggle/input/vgg_model_trained/keras/default/1/vgg_UF_model_20epoch.keras\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:29:44.656165Z","iopub.execute_input":"2024-09-30T04:29:44.656687Z","iopub.status.idle":"2024-09-30T04:30:06.803493Z","shell.execute_reply.started":"2024-09-30T04:29:44.656637Z","shell.execute_reply":"2024-09-30T04:30:06.802156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\n# Initialize results storage\nresults = {\n    'row_id': [],\n    'normal_mild': [],\n    'moderate': [],\n    'severe': []\n}\n\nbatch_size = 16  \n\n# Use tqdm to create a progress bar for the entire dataset\nwith tqdm(total=len(test_dataset), desc=\"Processing images\") as pbar:\n    # Iterate over batches of data\n    for idx in range(len(test_dataset)):\n        try:\n            # Get a batch of images and corresponding row IDs\n            images, row_ids = test_dataset[idx]\n\n            # Ensure the images have the shape (batch_size, 256, 256, 3)\n            if images.shape[1:] != (256, 256, 3):\n                raise ValueError(f\"Image batch shape is {images.shape}, expected (?, 256, 256, 3)\")\n\n            # Make predictions on the batch with verbose=0 to suppress output\n            predictions = model.predict(images, verbose=0)  # Shape: (batch_size, num_classes)\n\n            # Append results for each image in the batch\n            for i in range(len(row_ids)):\n                probs = predictions[i]\n\n                results['row_id'].append(row_ids[i])\n                results['normal_mild'].append(probs[0])  # Class 0: Normal/Mild\n                results['moderate'].append(probs[1])     # Class 1: Moderate\n                results['severe'].append(probs[2])       # Class 2: Severe\n\n            # Update the progress bar by the batch size\n            pbar.update(1)\n\n        except Exception as e:\n            print(f\"Error processing index {idx}: {e}\")\n\n# Convert results to DataFrame\nresults_df = pd.DataFrame(results)\n\n# Normalize probabilities to ensure they sum to 1\nresults_df[['normal_mild', 'moderate', 'severe']] = results_df[['normal_mild', 'moderate', 'severe']].div(\n    results_df[['normal_mild', 'moderate', 'severe']].sum(axis=1), axis=0\n)\n\n# Save the results to a CSV file\nresults_df.to_csv('test_predictions.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:30:06.809704Z","iopub.execute_input":"2024-09-30T04:30:06.810984Z","iopub.status.idle":"2024-09-30T04:31:40.537887Z","shell.execute_reply.started":"2024-09-30T04:30:06.810893Z","shell.execute_reply":"2024-09-30T04:31:40.536407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:31:46.458857Z","iopub.execute_input":"2024-09-30T04:31:46.459343Z","iopub.status.idle":"2024-09-30T04:31:46.472966Z","shell.execute_reply.started":"2024-09-30T04:31:46.459297Z","shell.execute_reply":"2024-09-30T04:31:46.471549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Average results per row_id\naveraged_results_df = results_df.groupby('row_id', as_index=False).mean()\n\n# Normalize probabilities to ensure they sum to 1\nsum_probs = averaged_results_df[['normal_mild', 'moderate', 'severe']].sum(axis=1)\naveraged_results_df['normal_mild'] = averaged_results_df['normal_mild'] / sum_probs\naveraged_results_df['moderate'] = averaged_results_df['moderate'] / sum_probs\naveraged_results_df['severe'] = averaged_results_df['severe'] / sum_probs\n\n# Check for any invalid values\nif (averaged_results_df[['normal_mild', 'moderate', 'severe']] < 0).any().any():\n    raise ValueError(\"Found negative probabilities in submission.\")","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:31:47.389065Z","iopub.execute_input":"2024-09-30T04:31:47.389588Z","iopub.status.idle":"2024-09-30T04:31:47.408603Z","shell.execute_reply.started":"2024-09-30T04:31:47.389537Z","shell.execute_reply":"2024-09-30T04:31:47.407226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = averaged_results_df[['row_id', 'normal_mild', 'moderate', 'severe']]\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:31:48.042278Z","iopub.execute_input":"2024-09-30T04:31:48.042742Z","iopub.status.idle":"2024-09-30T04:31:48.065109Z","shell.execute_reply.started":"2024-09-30T04:31:48.042695Z","shell.execute_reply":"2024-09-30T04:31:48.063335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as 'submission.csv'.\")","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:31:48.759379Z","iopub.execute_input":"2024-09-30T04:31:48.759848Z","iopub.status.idle":"2024-09-30T04:31:48.769777Z","shell.execute_reply.started":"2024-09-30T04:31:48.759802Z","shell.execute_reply":"2024-09-30T04:31:48.768371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission file\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T04:31:49.950798Z","iopub.execute_input":"2024-09-30T04:31:49.951293Z","iopub.status.idle":"2024-09-30T04:31:49.959235Z","shell.execute_reply.started":"2024-09-30T04:31:49.951248Z","shell.execute_reply":"2024-09-30T04:31:49.957662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}