{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":8758964,"sourceType":"datasetVersion","datasetId":5262301},{"sourceId":8760417,"sourceType":"datasetVersion","datasetId":5263359},{"sourceId":184917418,"sourceType":"kernelVersion"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Configure notebook to use GP**\n\nEnsure that tensorflow uses GPU\n\n> Check if tensorflow is having access to GPU","metadata":{}},{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n#import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display\nimport logging\nfrom multiprocessing import Pool\nimport plotly.express as px #useful to display dicom image.\nimport tensorflow as tf\nimport pickle\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove warnings !\nimport warnings\nwarnings.filterwarnings('ignore')\n#Cinfgure gpus:\n\ngpus=tf.config.list_physical_devices('GPU')\nimport tensorflow as tf\nif (len(tf.config.list_physical_devices('GPU'))==1):\n    print(\"TensorFlow has access to the GPU.\")\n\n#Ensure that Tensorflow uses GPU\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n            print(\"GPU is available and configured sucesssfully\")\n    except RuntimeError as e:\n        print(e)\nelse:\n    print('No GPUS found!')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports functions from rsna_functions","metadata":{}},{"cell_type":"code","source":"%run \"/kaggle/usr/lib/rsna_functons/rsna_functons.py\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define/Asign file paths -:","metadata":{}},{"cell_type":"code","source":"\n# Assign file path to simple variables:\n# CSV fiels provided and thier paths:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\nsample_csv_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntrain_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\ntrain_label_coordinates_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntest_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n# Image folder paths:\ntrain_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\ntest_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'\n\n\n# Read csv files:\nsample_csv=pd.read_csv(sample_csv_path)\ntrain_data=pd.read_csv(train_data_path)\ntrain_series_description=pd.read_csv(train_series_descriptions_path)\ntest_series_description=pd.read_csv(test_series_descriptions_path)\ntrain_label_coordinates_data=pd.read_csv(train_label_coordinates_data_path)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add a new column to store the categories in the train_labels_coordinates_data dataset.\ntrain_label_coordinates_data[\"category\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    r = row['study_id']\n    col = (row['condition'].lower().replace(' ', '_')) + '_' + (row['level'].lower().replace('/', '_'))\n    \n    # Ensure the column exists in train_data to avoid KeyError\n    if col in train_data.columns:\n        # Check if there is a matching study_id and the column exists\n        value = train_data.loc[train_data['study_id'] == r, col].values\n        if len(value) > 0:\n            train_label_coordinates_data.at[idx, \"category\"] = value[0]\n        else:\n            train_label_coordinates_data.at[idx, \"category\"] = None\n    else:\n        train_label_coordinates_data.at[idx, \"category\"] = None\n\n# Display the first 5 rows of the updated DataFrame\ntrain_label_coordinates_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we need to create a list of image file paths :\n# For this, let us first create a new df containing only the first three columns. All we need to do is to drop rest of the columns.\nfirst_three_cols=train_label_coordinates_data.iloc[:,:3]\n\n\n# Now all we have to do is to convert this df into a new df that contains combined values of the each row.\njoin_row_as_str=lambda row: train_images_path+\"/\"+\"/\".join(row.astype(str))+\".dcm\"\ntrain_label_images_path=first_three_cols.apply(join_row_as_str,axis=1)\n\n#check for the integrity of the new df\nprint(f'The number of rows in train_label_images_path and train_label_coordinates_data are equal? {train_label_images_path.shape[0]==train_label_coordinates_data.shape[0]}')\n\n#Display the shape of the updated dataset:\ntrain_label_coordinates_data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add img file patth as column in train_label_coordinates_data\n# Add a new column to store the categories\ntrain_label_coordinates_data[\"img_file_path\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    file_path = train_label_images_path.loc[idx]\n    train_label_coordinates_data.at[idx, \"img_file_path\"] = file_path\n    \n# Display the first row of the updated DataFrame\ntrain_label_coordinates_data.head(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_label_coordinates_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\ndisplay(train_data.isnull().sum())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display rows where the \"category\" column has missing values\nmissing_category_entries = train_data[train_data['category'].isnull()]\n\ndisplay(missing_category_entries.head(2))\nprint(f\"Number of rows with missing 'category' entries: {len(missing_category_entries)}\\n----\\n\")\n\n# Get unique study IDs where 'category' is null\nunique_study_ids_missing_category_entries = missing_category_entries['study_id'].unique()\n\n# Count the number of unique study IDs\nnum_unique_study_ids_missing_category_entries = len(unique_study_ids_missing_category_entries)\n\n# Print the number of unique study IDs with missing category entries\nprint(f\"Number of unique study IDs with missing 'category' entries: {num_unique_study_ids_missing_category_entries}\")\ndisplay(unique_study_ids_missing_category_entries)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop rows with missing values without modifiying original df (hence inplace=False) and store this in new df names train:\ntrain=train_data.dropna(inplace=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confirm if there are still any missing valyes in new df:\nnum_of_missing_values=train.isnull().sum()\nprint(num_of_missing_values)\n\n#Visualise the original and new df shapes and compare them :\nprint(f'\\n----\\nThe original dataframe (train_data) had {train_data.shape[0]} rows and {train_data.shape[1]} columns.')\nprint(f'The new dataframe (train) has {train.shape[0]} rows and {train.shape[1]} columns\\n----\\n')\nprint(f'We succesfulyl dropped {train_data.shape[0] - train.shape[0]} rows and {num_unique_study_ids_missing_category_entries} patients.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Split the data\ntrain_data, val_data = train_test_split(train, test_size=0.2, stratify=train_data['category'], random_state=42)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Output the DataFrames to CSV files\ntrain_data.to_csv('train_data.csv', index=False)\nval_data.to_csv('val_data.csv', index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nimport os\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Encode the labels\nlabel_encoder = LabelEncoder()\ntrain_data['category'] = label_encoder.fit_transform(train_data['category'])\nval_data['category'] = label_encoder.transform(val_data['category'])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport pydicom\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nclass DicomDataGenerator(Sequence):\n    def __init__(self, dataframe, x_col, y_col, batch_size, target_size, shuffle=True, augment=False):\n        self.dataframe = dataframe\n        self.x_col = x_col\n        self.y_col = y_col\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n        \n        self.datagen = ImageDataGenerator(\n            rotation_range=20,\n            zoom_range=0.15,\n            width_shift_range=0.2,\n            height_shift_range=0.2,\n            shear_range=0.15,\n            horizontal_flip=True,\n            fill_mode=\"nearest\"\n        )\n\n    def __len__(self):\n        return int(np.floor(len(self.dataframe) / self.batch_size))\n\n    def __getitem__(self, index):\n        batch = self.dataframe.iloc[index*self.batch_size:(index+1)*self.batch_size]\n        x, y = self.__data_generation(batch)\n        return x, y\n\n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.dataframe))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, batch):\n        images = []\n        labels = []\n\n        for _, row in batch.iterrows():\n            dicom_path = row[self.x_col]\n            dicom = pydicom.dcmread(dicom_path)\n            image = dicom.pixel_array\n            image = cv2.resize(image, self.target_size)\n            image = cv2.normalize(image, None, 0, 255, cv2.NORM_MINMAX)\n            image = image.astype('float32') / 255.0\n            image = np.expand_dims(image, axis=-1)\n            if self.augment:\n                image = self.datagen.random_transform(image)\n            images.append(image)\n            labels.append(row[self.y_col])\n        \n        x = np.array(images)\n        y = tf.keras.utils.to_categorical(labels, num_classes=3)\n        return x, y\n\n# Assuming train_data and val_data are your DataFrames with 'img_file_path' and 'category' columns\ntrain_data['category'] = train_data['category'].astype(int)\nval_data['category'] = val_data['category'].astype(int)\n\n# Create data generators\ntrain_generator = DicomDataGenerator(\n    dataframe=train_data,\n    x_col='img_file_path',\n    y_col='category',\n    batch_size=32,\n    target_size=(224, 224),\n    shuffle=True,\n    augment=True\n)\n\nval_generator = DicomDataGenerator(\n    dataframe=val_data,\n    x_col='img_file_path',\n    y_col='category',\n    batch_size=32,\n    target_size=(224, 224),\n    shuffle=False,\n    augment=False\n)\n\n# Model Architecture\nmodel = tf.keras.Sequential([\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 1)),  # Assuming grayscale images\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.Dense(3, activation='softmax')  # Adjust the output layer based on the number of classes\n])\n\n# Compile the Model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the Model\nhistory = model.fit(train_generator, epochs=50, validation_data=val_generator)\n\n# Evaluate the Model\nval_loss, val_accuracy = model.evaluate(val_generator)\nprint(f'Validation Accuracy: {val_accuracy}')\n\n# Create Predictions for Submission\ntest_data['img_file_path'] = test_data['img_file_path'].astype(str)\ntest_data['category'] = test_data['category'].astype(int)  # Ensure categories are correct\n\ntest_generator = DicomDataGenerator(\n    dataframe=test_data,\n    x_col='img_file_path',\n    y_col='category',\n    batch_size=32,\n    target_size=(224, 224),\n    shuffle=False,\n    augment=False\n)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_generator)\npredictions_df = pd.DataFrame(predictions, columns=['normal_mild', 'moderate', 'severe'])\n\n# Add row_id for submission\ntest_data['row_id'] = test_data.apply(lambda x: f\"{x['study_id']}_{x['condition'].replace(' ', '_')}_{x['level'].replace('/', '_')}\", axis=1)\nsubmission = pd.concat([test_data['row_id'], predictions_df], axis=1)\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}