{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"o create a model for detecting and classifying degenerative spine conditions using lumbar spine MR images, we'll need to develop a deep learning approach. Here's an outline of the steps we'll follow to build this model using Python:\n\n    Data Preprocessing:\n        Load and preprocess MRI images.\n        Normalize the images.\n        Split the dataset into training, validation, and test sets.\n\n    Model Architecture:\n        Use a Convolutional Neural Network (CNN) architecture for image classification.\n        Implement transfer learning using a pre-trained model (e.g., ResNet50, VGG16) to leverage pre-learned features.\n\n    Training the Model:\n        Define the loss function and optimizer.\n        Train the model on the training data and validate it on the validation set.\n        Use data augmentation to improve the robustness of the model.\n\n    Evaluation:\n        Evaluate the model on the test set using the sample weighted log loss metric.\n        Predict the severity scores for each condition and vertebral level.\n\n    Submission:\n        Format the predictions in the required submission format.\n\nHere is an example of how you could implement this in Python using TensorFlow and Keras:","metadata":{}},{"cell_type":"markdown","source":"**Step 1: Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n# Assuming data directory structure\ndata_dir = \"path/to/data\"\nlabels_file = \"path/to/labels.csv\"\n\n# Load the labels\nlabels_df = pd.read_csv(labels_file)\n\n# Data generator for loading and augmenting images\ndatagen = ImageDataGenerator(\n    rescale=1./255,\n    validation_split=0.2,  # 80-20 train-validation split\n    horizontal_flip=True,\n    vertical_flip=True\n)\n\ntrain_gen = datagen.flow_from_dataframe(\n    dataframe=labels_df,\n    directory=data_dir,\n    x_col=\"image_path\",\n    y_col=[\"left_neural_foraminal_narrowing\", \"right_neural_foraminal_narrowing\",\n           \"left_subarticular_stenosis\", \"right_subarticular_stenosis\", \"spinal_canal_stenosis\"],\n    subset=\"training\",\n    batch_size=32,\n    shuffle=True,\n    class_mode=\"raw\",\n    target_size=(224, 224)\n)\n\nval_gen = datagen.flow_from_dataframe(\n    dataframe=labels_df,\n    directory=data_dir,\n    x_col=\"image_path\",\n    y_col=[\"left_neural_foraminal_narrowing\", \"right_neural_foraminal_narrowing\",\n           \"left_subarticular_stenosis\", \"right_subarticular_stenosis\", \"spinal_canal_stenosis\"],\n    subset=\"validation\",\n    batch_size=32,\n    shuffle=True,\n    class_mode=\"raw\",\n    target_size=(224, 224)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-31T08:00:55.605207Z","iopub.execute_input":"2024-05-31T08:00:55.605637Z","iopub.status.idle":"2024-05-31T08:01:15.135872Z","shell.execute_reply.started":"2024-05-31T08:00:55.605596Z","shell.execute_reply":"2024-05-31T08:01:15.13386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Step 2: Model Architecture**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Model\n\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# Add custom layers on top of ResNet50\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.5)(x)\noutput = Dense(15, activation='softmax')(x)  # 5 conditions * 3 severity levels\n\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Freeze the base model layers\nfor layer in base_model.layers:\n    layer.trainable = False\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Step 3: Training the Model**","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    steps_per_epoch=train_gen.samples // train_gen.batch_size,\n    validation_data=val_gen,\n    validation_steps=val_gen.samples // val_gen.batch_size,\n    epochs=10\n)\n\n# Unfreeze some layers and fine-tune\nfor layer in base_model.layers[-10:]:\n    layer.trainable = True\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss='categorical_crossentropy', metrics=['accuracy'])\n\nhistory_fine = model.fit(\n    train_gen,\n    steps_per_epoch=train_gen.samples // train_gen.batch_size,\n    validation_data=val_gen,\n    validation_steps=val_gen.samples // val_gen.batch_size,\n    epochs=10\n)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Step 4: Evaluation**","metadata":{}},{"cell_type":"code","source":"test_gen = ImageDataGenerator(rescale=1./255).flow_from_dataframe(\n    dataframe=labels_df,  # This should be a test set dataframe\n    directory=data_dir,\n    x_col=\"image_path\",\n    y_col=None,\n    batch_size=32,\n    shuffle=False,\n    class_mode=None,\n    target_size=(224, 224)\n)\n\npreds = model.predict(test_gen, steps=test_gen.samples // test_gen.batch_size + 1)\npred_labels = np.argmax(preds, axis=1)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Step 5: Submission**","metadata":{}},{"cell_type":"code","source":"# Format the predictions in the required submission format\nsubmission = pd.DataFrame({\n    \"row_id\": test_gen.filenames,\n    \"normal_mild\": preds[:, 0],\n    \"moderate\": preds[:, 1],\n    \"severe\": preds[:, 2]\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}