{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport cv2 as cv\nfrom path import Path\nimport os \nimport glob\nimport tensorflow_hub as hub\nimport os \nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import layers\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import applications\nimport albumentations as A\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-06T12:15:35.126322Z","iopub.execute_input":"2023-06-06T12:15:35.126585Z","iopub.status.idle":"2023-06-06T12:15:45.907558Z","shell.execute_reply.started":"2023-06-06T12:15:35.12656Z","shell.execute_reply":"2023-06-06T12:15:45.90661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df= pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nsample_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:15:45.910096Z","iopub.execute_input":"2023-06-06T12:15:45.910782Z","iopub.status.idle":"2023-06-06T12:15:45.932247Z","shell.execute_reply.started":"2023-06-06T12:15:45.910746Z","shell.execute_reply":"2023-06-06T12:15:45.931411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    dicom=pydicom.read_file(path)\n    data=dicom.pixel_array\n    data=data-np.min(data)\n    if np.max(data) != 0:\n        data=data/np.max(data)\n    data=(data*255).astype(np.uint8)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:15:45.933601Z","iopub.execute_input":"2023-06-06T12:15:45.933955Z","iopub.status.idle":"2023-06-06T12:15:45.941686Z","shell.execute_reply.started":"2023-06-06T12:15:45.933924Z","shell.execute_reply":"2023-06-06T12:15:45.940752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'\ntrainset = []\ntrainlabel = []\ntrainidt = []\n\n# Filter out samples with BraTS21ID [00109, 00123, 00709]\nfiltered_train_df = train_df.loc[~train_df['BraTS21ID'].isin([109, 123, 709])]\n\nfor _, row in tqdm(filtered_train_df.iterrows(), total=len(filtered_train_df)):\n    idt = row['BraTS21ID']\n    idt2 = ('00000' + str(idt))[-5:]\n    path = os.path.join(train_dir, idt2, 'T2w')\n\n    for im in os.listdir(path):\n        img = load_dicom(os.path.join(path, im))\n        img = cv.resize(img, (64, 64))\n        \n        image = img_to_array(img)\n        image = image / 255.0\n        \n        # Skip black images with 0 intensities\n        if np.sum(image) == 0:\n            continue\n\n        trainset.append(image)\n        trainlabel.append(row['MGMT_value'])\n        trainidt.append(idt)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:15:45.944548Z","iopub.execute_input":"2023-06-06T12:15:45.944915Z","iopub.status.idle":"2023-06-06T12:32:50.463859Z","shell.execute_reply.started":"2023-06-06T12:15:45.944883Z","shell.execute_reply":"2023-06-06T12:32:50.462819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir='../input/rsna-miccai-brain-tumor-radiogenomic-classification/test'\ntestset=[]\ntestidt=[]\nfor i in tqdm(range(len(sample_df))):\n    idt=sample_df.loc[i,'BraTS21ID']\n    idt2=('00000'+str(idt))[-5:]\n    path=os.path.join(test_dir,idt2,'T2w')               \n    for im in os.listdir(path):   \n        img=load_dicom(os.path.join(path,im))\n        img=cv.resize(img,(64,64)) \n        image=img_to_array(img)\n        image=image/255.0\n        testset+=[image]\n        testidt+=[idt]","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:32:50.465383Z","iopub.execute_input":"2023-06-06T12:32:50.465992Z","iopub.status.idle":"2023-06-06T12:35:13.572419Z","shell.execute_reply.started":"2023-06-06T12:32:50.465957Z","shell.execute_reply":"2023-06-06T12:35:13.57148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=np.array(trainlabel)\nY_train=to_categorical(y)\nX_train=np.array(trainset)\nX_test=np.array(testset)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:00.235888Z","iopub.execute_input":"2023-06-06T14:10:00.236271Z","iopub.status.idle":"2023-06-06T14:10:02.045096Z","shell.execute_reply.started":"2023-06-06T14:10:00.23624Z","shell.execute_reply":"2023-06-06T14:10:02.044079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, Y_train, Y_val = train_test_split(X_train, Y_train, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:02.046997Z","iopub.execute_input":"2023-06-06T14:10:02.047358Z","iopub.status.idle":"2023-06-06T14:10:03.353907Z","shell.execute_reply.started":"2023-06-06T14:10:02.047324Z","shell.execute_reply":"2023-06-06T14:10:03.352959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the BraTS21ID and MGMT_value for the validation set\nvalidation_idt = [trainidt[i] for i in range(len(Y_val))]\nvalidation_label = [trainlabel[i] for i in range(len(Y_val))]\n\n# Convert validation data to a pandas DataFrame\nvalidation_data = {\n    'BraTS21ID': validation_idt,\n    'MGMT_value': validation_label\n}\nvalidation_df = pd.DataFrame(validation_data)\n\n# Save validation data to a CSV file\nvalidation_df.to_csv('validation.csv', index=False)\nvalidation_df\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:16.838422Z","iopub.execute_input":"2023-06-06T12:35:16.838835Z","iopub.status.idle":"2023-06-06T12:35:17.028258Z","shell.execute_reply.started":"2023-06-06T12:35:16.838796Z","shell.execute_reply":"2023-06-06T12:35:17.027377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group the actual validation data by 'BraTS21ID' and calculate the mean\nactual_val_grouped = validation_df.groupby('BraTS21ID')['MGMT_value'].mean().reset_index()\nactual_val_grouped['MGMT_value'] = actual_val_grouped['MGMT_value'].apply(lambda x: 1 if x >= 0.5 else 0)\n\nactual_val_grouped\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:17.029706Z","iopub.execute_input":"2023-06-06T12:35:17.030046Z","iopub.status.idle":"2023-06-06T12:35:17.052061Z","shell.execute_reply.started":"2023-06-06T12:35:17.030012Z","shell.execute_reply":"2023-06-06T12:35:17.050882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual_val_grouped.to_csv('Actual_Validation_Grouped.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:17.058476Z","iopub.execute_input":"2023-06-06T12:35:17.058775Z","iopub.status.idle":"2023-06-06T12:35:17.065141Z","shell.execute_reply.started":"2023-06-06T12:35:17.05875Z","shell.execute_reply":"2023-06-06T12:35:17.06366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height,img_width = 64,64 \nnum_classes = 2\nbase_model = applications.resnet.ResNet50(weights= None, include_top=False, input_shape= (img_height,img_width,1))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:17.067341Z","iopub.execute_input":"2023-06-06T12:35:17.068114Z","iopub.status.idle":"2023-06-06T12:35:21.350214Z","shell.execute_reply.started":"2023-06-06T12:35:17.06808Z","shell.execute_reply":"2023-06-06T12:35:21.349224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nx = base_model.output\nx = keras.layers.GlobalAveragePooling2D()(x)\nx = keras.layers.Dropout(0.5)(x)\npredictions = keras.layers.Dense(num_classes, activation= 'sigmoid')(x)\nmodel = keras.models.Model(inputs = base_model.input, outputs = predictions)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:21.351566Z","iopub.execute_input":"2023-06-06T12:35:21.351938Z","iopub.status.idle":"2023-06-06T12:35:21.752283Z","shell.execute_reply.started":"2023-06-06T12:35:21.351905Z","shell.execute_reply":"2023-06-06T12:35:21.751549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\noptimizer = Adam(learning_rate=1e-4)\n\n# Compile the model\nmodel.compile(optimizer=optimizer,\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=False),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:21.753316Z","iopub.execute_input":"2023-06-06T12:35:21.753854Z","iopub.status.idle":"2023-06-06T12:35:21.807368Z","shell.execute_reply.started":"2023-06-06T12:35:21.753818Z","shell.execute_reply":"2023-06-06T12:35:21.806625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, Y_train, epochs=100, batch_size=64, validation_data=(X_val, Y_val))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T12:35:21.808554Z","iopub.execute_input":"2023-06-06T12:35:21.808885Z","iopub.status.idle":"2023-06-06T14:05:01.027185Z","shell.execute_reply.started":"2023-06-06T12:35:21.808854Z","shell.execute_reply":"2023-06-06T14:05:01.026192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training loss (excluding epoch 21)\nplt.plot(history.history['loss'] , label='Training Loss')\n# Plot validation loss (excluding epoch 21)\nplt.plot(history.history['val_loss'], label='Validation Loss')\n\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Plot training accuracy (excluding epoch 21)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\n# Plot validation accuracy (excluding epoch 21)\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:05:01.029043Z","iopub.execute_input":"2023-06-06T14:05:01.029417Z","iopub.status.idle":"2023-06-06T14:05:01.596511Z","shell.execute_reply.started":"2023-06-06T14:05:01.029381Z","shell.execute_reply":"2023-06-06T14:05:01.595566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(X_test)\npred=np.argmax(y_pred,axis=1)\nresult=pd.DataFrame(testidt)\nresult[1]=pred\nresult.columns=['BraTS21ID','MGMT_value']\nresult2=result.groupby('BraTS21ID',as_index=False).mean()\nresult2.to_csv('ResNet50_T2w_test.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:05:01.598236Z","iopub.execute_input":"2023-06-06T14:05:01.598932Z","iopub.status.idle":"2023-06-06T14:05:08.944873Z","shell.execute_reply.started":"2023-06-06T14:05:01.598896Z","shell.execute_reply":"2023-06-06T14:05:08.943748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(X_val)\npred=np.argmax(y_pred,axis=1)\nresult=pd.DataFrame(validation_idt)\nresult[1]=pred\nresult.columns=['BraTS21ID','MGMT_value']\nresult2=result.groupby('BraTS21ID',as_index=False).mean()\nresult2['MGMT_value']=result2['MGMT_value'].apply(lambda x: 1 if x>=0.5 else 0)\nresult2.to_csv('ResNet50_T2w_valid.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:05:08.946584Z","iopub.execute_input":"2023-06-06T14:05:08.946991Z","iopub.status.idle":"2023-06-06T14:05:14.564688Z","shell.execute_reply.started":"2023-06-06T14:05:08.946959Z","shell.execute_reply":"2023-06-06T14:05:14.563591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\n# Perform manual prediction on the validation set\ny_val_pred = model.predict(X_val)\nval_pred = (y_val_pred > 0.5).astype(int)\n\n# Calculate and print classification report\ntarget_names = ['0', '1']\nprint(classification_report(Y_val, val_pred, target_names=target_names))\n# Convert multilabel-indicator format to single-label format\nY_val = Y_val.argmax(axis=1)\nval_pred = val_pred.argmax(axis=1)\n# Calculate and plot confusion matrix\ncm = confusion_matrix(Y_val, val_pred)\ncm = pd.DataFrame(cm, index=target_names, columns=target_names)\nplt.figure(figsize=(10, 10))\nsns.heatmap(cm, cmap=\"YlOrRd\", linecolor='white', linewidth=1, annot=True, fmt='', xticklabels=target_names, yticklabels=target_names)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:05:14.566231Z","iopub.execute_input":"2023-06-06T14:05:14.566561Z","iopub.status.idle":"2023-06-06T14:05:21.128852Z","shell.execute_reply.started":"2023-06-06T14:05:14.566528Z","shell.execute_reply":"2023-06-06T14:05:21.1278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Save the trained model to a file\nmodel.save('/kaggle/working/ResNet50_T2w.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:05:21.130238Z","iopub.execute_input":"2023-06-06T14:05:21.130704Z","iopub.status.idle":"2023-06-06T14:05:22.478839Z","shell.execute_reply.started":"2023-06-06T14:05:21.130668Z","shell.execute_reply":"2023-06-06T14:05:22.477893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimg_height, img_width = 64, 64\nnum_classes = 2\n\nbase_model = applications.efficientnet.EfficientNetB0(weights= None, include_top=False, input_shape= (img_height,img_width,1))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:37.868293Z","iopub.execute_input":"2023-06-06T14:10:37.868684Z","iopub.status.idle":"2023-06-06T14:10:39.487858Z","shell.execute_reply.started":"2023-06-06T14:10:37.868624Z","shell.execute_reply":"2023-06-06T14:10:39.486909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = base_model.output\nx = keras.layers.GlobalAveragePooling2D()(x)\nx = keras.layers.Dropout(0.5)(x)\npredictions = keras.layers.Dense(num_classes, activation= 'sigmoid')(x)\nmodel = keras.models.Model(inputs = base_model.input, outputs = predictions)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:39.489817Z","iopub.execute_input":"2023-06-06T14:10:39.490153Z","iopub.status.idle":"2023-06-06T14:10:40.018605Z","shell.execute_reply.started":"2023-06-06T14:10:39.490121Z","shell.execute_reply":"2023-06-06T14:10:40.017824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\noptimizer = Adam(learning_rate=1e-4)\n\n# Compile the model\nmodel.compile(optimizer=optimizer,\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=False),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:40.019688Z","iopub.execute_input":"2023-06-06T14:10:40.020032Z","iopub.status.idle":"2023-06-06T14:10:40.063243Z","shell.execute_reply.started":"2023-06-06T14:10:40.019998Z","shell.execute_reply":"2023-06-06T14:10:40.062465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, Y_train, epochs=100, batch_size=64, validation_data=(X_val, Y_val))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T14:10:40.065224Z","iopub.execute_input":"2023-06-06T14:10:40.065549Z","iopub.status.idle":"2023-06-06T15:31:22.49928Z","shell.execute_reply.started":"2023-06-06T14:10:40.065517Z","shell.execute_reply":"2023-06-06T15:31:22.498223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training loss\nplt.plot(history.history['loss'], label='Training Loss')\n# Plot validation loss\nplt.plot(history.history['val_loss'], label='Validation Loss')\n\n\nplt.xlabel('Epochs')\nplt.ylabel('Metrics')\nplt.legend()\nplt.show()\n\n# Plot training accuracy\nplt.plot(history.history['accuracy'], label='Training Accuracy')\n# Plot validation accuracy\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n\nplt.xlabel('Epochs')\nplt.ylabel('Metrics')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T15:31:22.500844Z","iopub.execute_input":"2023-06-06T15:31:22.501301Z","iopub.status.idle":"2023-06-06T15:31:23.028723Z","shell.execute_reply.started":"2023-06-06T15:31:22.501264Z","shell.execute_reply":"2023-06-06T15:31:23.027827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(X_test)\npred=np.argmax(y_pred,axis=1)\nresult=pd.DataFrame(testidt)\nresult[1]=pred\nresult.columns=['BraTS21ID','MGMT_value']\nresult2=result.groupby('BraTS21ID',as_index=False).mean()\nresult2.to_csv('EfficientNetV2_T2w_test.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2023-06-06T15:31:23.030375Z","iopub.execute_input":"2023-06-06T15:31:23.031077Z","iopub.status.idle":"2023-06-06T15:31:30.273075Z","shell.execute_reply.started":"2023-06-06T15:31:23.03104Z","shell.execute_reply":"2023-06-06T15:31:30.272149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(X_val)\npred=np.argmax(y_pred,axis=1)\nresult=pd.DataFrame(validation_idt)\nresult[1]=pred\nresult.columns=['BraTS21ID','MGMT_value']\nresult2=result.groupby('BraTS21ID',as_index=False).mean()\nresult2['MGMT_value']=result2['MGMT_value'].apply(lambda x: 1 if x>=0.5 else 0)\nresult2.to_csv('EfficientNetV2_T2w_valid.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2023-06-06T15:31:30.275505Z","iopub.execute_input":"2023-06-06T15:31:30.275956Z","iopub.status.idle":"2023-06-06T15:31:35.981971Z","shell.execute_reply.started":"2023-06-06T15:31:30.275922Z","shell.execute_reply":"2023-06-06T15:31:35.980783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\n\n# Perform manual prediction on the validation set\ny_val_pred = model.predict(X_val)\nval_pred = (y_val_pred > 0.5).astype(int)\n\n# Calculate and print classification report\ntarget_names = ['0', '1']\nprint(classification_report(Y_val, val_pred, target_names=target_names))\n# Convert multilabel-indicator format to single-label format\nY_val = Y_val.argmax(axis=1)\nval_pred = val_pred.argmax(axis=1)\n# Calculate and plot confusion matrix\ncm = confusion_matrix(Y_val, val_pred)\ncm = pd.DataFrame(cm, index=target_names, columns=target_names)\nplt.figure(figsize=(10, 10))\nsns.heatmap(cm, cmap=\"YlOrRd\", linecolor='white', linewidth=1, annot=True, fmt='', xticklabels=target_names, yticklabels=target_names)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T15:31:35.983543Z","iopub.execute_input":"2023-06-06T15:31:35.98392Z","iopub.status.idle":"2023-06-06T15:31:41.044461Z","shell.execute_reply.started":"2023-06-06T15:31:35.983887Z","shell.execute_reply":"2023-06-06T15:31:41.043537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the trained model to a file\nmodel.save('/kaggle/working/EfficientNetV2_T2w.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T15:31:41.046083Z","iopub.execute_input":"2023-06-06T15:31:41.046448Z","iopub.status.idle":"2023-06-06T15:31:41.855573Z","shell.execute_reply.started":"2023-06-06T15:31:41.046411Z","shell.execute_reply":"2023-06-06T15:31:41.854602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}