{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":8768680,"sourceType":"datasetVersion","datasetId":5269188},{"sourceId":184495011,"sourceType":"kernelVersion"},{"sourceId":184917418,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\nimport os as os\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display\nimport logging\nfrom multiprocessing import Pool\nimport bz2\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nimport os\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:24.183187Z","iopub.execute_input":"2024-07-02T20:25:24.183618Z","iopub.status.idle":"2024-07-02T20:25:25.196837Z","shell.execute_reply.started":"2024-07-02T20:25:24.183567Z","shell.execute_reply":"2024-07-02T20:25:25.196021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%run \"/kaggle/usr/lib/rsna_functons/rsna_functons.py\"\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:25.198679Z","iopub.execute_input":"2024-07-02T20:25:25.198972Z","iopub.status.idle":"2024-07-02T20:25:25.217312Z","shell.execute_reply.started":"2024-07-02T20:25:25.198946Z","shell.execute_reply":"2024-07-02T20:25:25.216327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the train and validation data.\ntrain_data=pd.read_csv(\"/kaggle/input/rsna-training-datasets/train_data.csv\")\nval_data=pd.read_csv(\"/kaggle/input/rsna-training-datasets/val_data.csv\")\n#Print statements to confirm sucessfull data loading.\nprint(f\"The training data loaded sucessfully. The train data has {train_data.shape[0]} rows and {train_data.shape[1]} columns.\")\nprint(f\"The tvalidatin data loaded sucessfully. The validation data has {val_data.shape[0]} rows and {val_data.shape[1]} columns.\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:25.218419Z","iopub.execute_input":"2024-07-02T20:25:25.218676Z","iopub.status.idle":"2024-07-02T20:25:25.498849Z","shell.execute_reply.started":"2024-07-02T20:25:25.218654Z","shell.execute_reply":"2024-07-02T20:25:25.497852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a custom loss function. We shall inport backened functions from keras library.This loss function ensures that we fullfill \n# the requirment of this competition (that is to specific weights)\n\nimport tensorflow.keras.backend as K\ndef weighted_log_loss(y_true, y_pred):\n    class_weights = K.constant([1.0, 2.0, 4.0])\n    y_true = K.cast(y_true, y_pred.dtype)\n    weights = K.sum(y_true * class_weights, axis=-1)\n    loss = K.sum(y_true * K.log(y_pred + K.epsilon()), axis=-1)\n    weighted_loss = -weights * loss\n    return K.mean(weighted_loss)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:25.500083Z","iopub.execute_input":"2024-07-02T20:25:25.500742Z","iopub.status.idle":"2024-07-02T20:25:25.508111Z","shell.execute_reply.started":"2024-07-02T20:25:25.500708Z","shell.execute_reply":"2024-07-02T20:25:25.507178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will create custom data generator because the pakaged imagedata generator which comes with keras do not have option of creating check points.\n# out data is big and by adding check points we will be saivng the progress of the model everytime it runs even if it fails to rune all the epochs (due various reasons such internet failure)\n\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nclass DicomDataGenerator(Sequence):\n    def __init__(self, dataframe, x_col, y_col, batch_size, target_size, shuffle=True, augment=False):\n        self.dataframe = dataframe\n        self.x_col = x_col\n        self.y_col = y_col\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n        \n        self.datagen = ImageDataGenerator(\n            rotation_range=20,\n            zoom_range=0.15,\n            width_shift_range=0.2,\n            height_shift_range=0.2,\n            shear_range=0.15,\n            horizontal_flip=True,\n            fill_mode=\"nearest\"\n        )\n\n    def __len__(self):\n        return int(np.floor(len(self.dataframe) / self.batch_size))\n\n    def __getitem__(self, index):\n        batch = self.dataframe.iloc[index*self.batch_size:(index+1)*self.batch_size]\n        x, y = self.__data_generation(batch)\n        return x, y\n\n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.dataframe))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, batch):\n        images = []\n        labels = []\n\n        for _, row in batch.iterrows():\n            dicom_path = row[self.x_col]\n            dicom = pydicom.dcmread(dicom_path)\n            image = dicom.pixel_array\n            image = cv2.resize(image, self.target_size)\n            image = cv2.normalize(image, None, 0, 255, cv2.NORM_MINMAX)\n            image = image.astype('float32') / 255.0\n            image = np.expand_dims(image, axis=-1)\n            if self.augment:\n                image = self.datagen.random_transform(image)\n            images.append(image)\n            labels.append(row[self.y_col])\n        \n        x = np.array(images)\n        y = tf.keras.utils.to_categorical(labels, num_classes=3)\n        return x, y\n\n# Assuming train_data and val_data are your DataFrames with 'img_file_path' and 'category' columns\ntrain_data['category'] = train_data['category'].astype(int)\nval_data['category'] = val_data['category'].astype(int)\n\n# Create data generators\ntrain_generator = DicomDataGenerator(\n    dataframe=train_data,\n    x_col='img_file_path',\n    y_col='category',\n    batch_size=32,\n    target_size=(224, 224),\n    shuffle=True,\n    augment=True\n)\n\nval_generator = DicomDataGenerator(\n    dataframe=val_data,\n    x_col='img_file_path',\n    y_col='category',\n    batch_size=32,\n    target_size=(224, 224),\n    shuffle=False,\n    augment=False\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:25.510439Z","iopub.execute_input":"2024-07-02T20:25:25.510709Z","iopub.status.idle":"2024-07-02T20:25:25.539469Z","shell.execute_reply.started":"2024-07-02T20:25:25.510686Z","shell.execute_reply":"2024-07-02T20:25:25.538447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Buidling model.\n\nfrom tensorflow.keras.layers import Input\n\n# Model Architecture\nmodel = tf.keras.Sequential([\n    Input(shape=(224, 224, 1)),  # Define input shape here\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.Dense(3, activation='softmax')  # Adjust the output layer based on the number of classes\n])\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:25.541176Z","iopub.execute_input":"2024-07-02T20:25:25.541583Z","iopub.status.idle":"2024-07-02T20:25:26.276169Z","shell.execute_reply.started":"2024-07-02T20:25:25.541548Z","shell.execute_reply":"2024-07-02T20:25:26.275190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, LambdaCallback\nfrom tensorflow.keras.models import load_model\n\n# Define the checkpoint callback for best loss\ncheckpoint_loss = ModelCheckpoint('model_checkpoint_best_loss.weights.h5', \n                                  save_best_only=True, save_weights_only=True, \n                                  monitor='val_loss', mode='min', verbose=1)\n\n# Define the checkpoint callback for best accuracy\ncheckpoint_accuracy = ModelCheckpoint('model_checkpoint_best_accuracy.weights.h5', \n                                      save_best_only=True, save_weights_only=True, \n                                      monitor='val_accuracy', mode='max', verbose=1)\n\n# Save model after each epoch\nsave_after_epoch = LambdaCallback(on_epoch_end=lambda epoch, logs: model.save('model_checkpoint_epoch.h5'))\n\n# Compile the Model with the custom weighted log loss\nmodel.compile(optimizer='adam', loss=weighted_log_loss, metrics=['accuracy'])\n\n# Load the last saved weights if they exist\ntry:\n    model.load_weights('model_checkpoint_epoch.h5')\n    print(\"Model weights loaded from the last checkpoint.\")\nexcept:\n    try:\n        model.load_weights('model_checkpoint_best_loss.weights.h5')\n        print(\"Model weights loaded from best validation loss checkpoint.\")\n    except:\n        try:\n            model.load_weights('model_checkpoint_best_accuracy.weights.h5')\n            print(\"Model weights loaded from best validation accuracy checkpoint.\")\n        except:\n            print(\"No checkpoint found. Starting training from scratch.\")\n\n# Train the model with both checkpoint callbacks\nhistory = model.fit(train_generator, epochs=50, validation_data=val_generator, \n                    callbacks=[checkpoint_loss, checkpoint_accuracy, save_after_epoch])\n\n# Save the fully trained model\nmodel.save('my_trained_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:25:26.277509Z","iopub.execute_input":"2024-07-02T20:25:26.277857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the Model\nval_loss, val_accuracy = model.evaluate(val_generator)\nprint(f'Validation Accuracy: {val_accuracy}')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Predictions for Submission\npredictions = model.predict(test_generator)\npredictions_df = pd.DataFrame(predictions, columns=['normal_mild', 'moderate', 'severe'])\n\n# Add row_id for submission\ntest_data['row_id'] = test_data.apply(lambda x: f\"{x['study_id']}_{x['condition'].replace(' ', '_')}_{x['level'].replace('/', '_')}\", axis=1)\nsubmission = pd.concat([test_data['row_id'], predictions_df], axis=1)\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{}}]}