{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from __future__ import print_function\nimport numpy as np\nimport warnings\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport os\nimport keras\n!pip install keras_applications\nfrom keras.models import Model\nfrom keras.layers import Flatten\nfrom keras.layers import Dense\nfrom keras.layers import Input\nfrom keras.layers import Conv2D\nfrom keras.layers import MaxPooling2D\nfrom keras.layers import GlobalMaxPooling2D\nfrom keras.layers import GlobalAveragePooling2D\nfrom keras.preprocessing import image\nfrom tensorflow.keras.utils import get_source_inputs \nfrom tensorflow.python.keras.utils import layer_utils \nfrom tensorflow.python.keras.utils.data_utils import get_file\nfrom keras import backend as K\nfrom tensorflow.keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nfrom sklearn.utils.class_weight import compute_class_weight\n\nfrom keras.applications.imagenet_utils import decode_predictions\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras_applications.imagenet_utils import _obtain_input_shape\nfrom tensorflow.keras.utils import get_source_inputs","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\n\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport os\n\n\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification'\ndataset_path = os.listdir(PATH)\n\nfolders = os.listdir(PATH)\nprint (folders)  #what kinds of files are in this dataset\n\nprint(\"No. of Files/Folders found: \", len(dataset_path))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(PATH+\"/train.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check how many samples for each category are present\nprint(\"Total number of data in the dataset: \", len(df))\n\ndata_count = df['expert_consensus'].value_counts()\n\nprint(\"expert_consensus data in each category: \")\nprint(data_count)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns=['eeg_sub_id', 'eeg_label_offset_seconds', \"spectrogram_sub_id\", \"spectrogram_label_offset_seconds\",\"label_id\",])\ndf=df.drop_duplicates(subset=(\"spectrogram_id\"))\ndata_count = df['expert_consensus'].value_counts()\n\nprint(\"expert_consensus data in each category: \")\nprint(data_count) ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.replace({'GRDA':'NO'}, inplace=True)\ndf.replace({'Other':'NO'}, inplace=True)\ndf.replace({'GPD':'NO'}, inplace=True)\ndf.replace({'LRDA':'NO'}, inplace=True)\ndf.replace({'LPD':'NO'}, inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder , OneHotEncoder\ny = df['expert_consensus'].values\nprint(y)\ny_labelencoder = LabelEncoder ()\ny = y_labelencoder.fit_transform (y)\nprint (y)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = y.reshape(-1, 1)  # Reshape to column vector\n# Create OneHotEncoder object with specified categories\nonehotencoder = OneHotEncoder(categories='auto')  # 'auto' automatically determines categories\nY = onehotencoder.fit_transform(y).toarray()\nprint(Y.shape)  # This will print the shape of the one-hot encoded output","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load spectrogram data from Parquet file\necg_spectrogram = pd.read_parquet(PATH+'/train_spectrograms/353733.parquet')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nfrom tensorflow.keras.preprocessing.image import img_to_array, array_to_img\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\n\n# Resizing and Preprocessing Spectrogram Images\ndef preprocess_spectrogram(spectrogram, target_size=(224, 224)):\n    # Normalizing spectrogram data to [0, 255] range \n    if spectrogram.dtype == np.float64:\n        spectrogram = (spectrogram - spectrogram.min()) / (spectrogram.max() - spectrogram.min()) * 255.0\n        spectrogram = spectrogram.astype(np.uint8)\n\n    # Resize spectrogram to target size using OpenCV\n    resized_spectrogram = cv2.resize(spectrogram, target_size)\n    # Convert to RGB image (3 channels)\n    rgb_spectrogram = cv2.cvtColor(resized_spectrogram, cv2.COLOR_GRAY2RGB)\n    # Convert to array and preprocess according to VGG-16 requirements\n    preprocessed_spectrogram = img_to_array(rgb_spectrogram)\n    preprocessed_spectrogram = preprocess_input(preprocessed_spectrogram)\n    return preprocessed_spectrogram\n\n# Preprocess first spectrogram in the DataFrame\nfirst_spectrogram = ecg_spectrogram.iloc[0, :].values  #each row represents a spectrogram\npreprocessed_spectrogram = preprocess_spectrogram(first_spectrogram)\nprint(preprocessed_spectrogram.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\npath='/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nimages=[]\n# Ignore all the warnings temporarily\nwith warnings.catch_warnings():\n    warnings.simplefilter(\"ignore\")\n    for f in os.listdir(path):\n        file_path=path+f\n        df = pd.read_parquet(file_path)\n        first_spectrogram = df.iloc[0, :].values  # Assuming each row represents a spectrogram\n        preprocessed_spectrogram = preprocess_spectrogram(first_spectrogram)\n        images.append(preprocessed_spectrogram)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = np.array(images)\nimages.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\n\n\nimages, Y = shuffle(images, Y, random_state=1)\n\ntrain_x, test_x, train_y, test_y = train_test_split(images, Y, test_size=0.2, random_state=128)\n\n#inspect the shape of the training and testing.\nprint(train_x.shape)\nprint(train_y.shape)\nprint(test_x.shape)\nprint(test_y.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers\nfrom tensorflow.keras.applications import EfficientNetB0\n\nNUM_CLASSES = 2\nIMG_SIZE = 224\nsize = (IMG_SIZE, IMG_SIZE)\n\n\ninputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n\n\n# Using model without transfer learning\n\noutputs = EfficientNetB0(include_top=True, weights=None, classes=NUM_CLASSES)(inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Model(inputs, outputs)\nmodel.compile(optimizer=\"adam\", loss=\"categorical_crossentropy\", metrics=[\"accuracy\"] )\n\nmodel.summary()\n\nhist=model.fit(train_x, train_y, epochs=200,validation_split=0.2, verbose=2, batch_size=16)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.evaluate(test_x, test_y)\nprint (\"Loss = \" + str(preds[0]))\nprint (\"Test Accuracy = \" + str(preds[1]))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n#Predictions on the test set\npred_y = model.predict(test_x)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the training loss at each epoch\nloss = hist.history['loss']\nepochs = range(1, len(loss) + 1)\nplt.figure(figsize=(8, 6))  \nplt.plot(epochs, loss, 'y')\nplt.title('Loss Curve')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the training accuracy at each epoch\nacc = hist.history['accuracy']\nplt.figure(figsize=(8, 6))\nplt.plot(epochs, acc, 'y')\nplt.title('Accuracy Curve')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score\nprecision = precision_score(test_y.argmax(axis=1), pred_y.argmax(axis=1), average='weighted')\nrecall = recall_score(test_y.argmax(axis=1),pred_y.argmax(axis=1), average='weighted')\nf1 = f1_score(test_y.argmax(axis=1),pred_y.argmax(axis=1), average='weighted')\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\nprint(\"F1 Score:\", f1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the training and validation loss at each epoch\nloss = hist.history['loss']\nval_loss = hist.history['val_loss']\nepochs = range(1, len(loss) + 1)\nplt.figure(figsize=(8, 6))  \nplt.plot(epochs, loss, 'y', label='Training Loss')\nplt.plot(epochs, val_loss, 'r', label='Validation Loss')\nplt.title('Loss Curve (Training and validation loss)')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the training and validation accuracy at each epoch\nacc = hist.history['accuracy']\nval_acc = hist.history['val_accuracy']\nplt.figure(figsize=(8, 6))\nplt.plot(epochs, acc, 'y', label='Training Accuracy')\nplt.plot(epochs, val_acc, 'r', label='Validation Accuracy')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(test_y.argmax(axis=1), pred_y.argmax(axis=1))\n\nprint(\"Confusion Matrix:\")\nprint(cm)","metadata":{},"execution_count":null,"outputs":[]}]}