{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":1493925,"sourceType":"datasetVersion","datasetId":877193},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":9841,"sourceType":"modelInstanceVersion","modelInstanceId":8013}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport sys\nimport pickle\nimport re\nimport os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nprint('TensorFlow version =',tf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-10T01:12:12.349996Z","iopub.execute_input":"2024-02-10T01:12:12.350443Z","iopub.status.idle":"2024-02-10T01:12:27.614726Z","shell.execute_reply.started":"2024-02-10T01:12:12.350416Z","shell.execute_reply":"2024-02-10T01:12:27.613806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preprocessing\nThis notebook is based off of Chris Deotte:\n- https://www.kaggle.com/code/cdeotte/efficientnetb0-starter-lb-0-43","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\n# absDimensions of the Dataset\nprint(f\"DF Shape: {df.shape}\")\nprint(f'Unique EEG: {df[\"eeg_id\"].nunique()}')\nprint(f'Unique Spectrogram: {df[\"spectrogram_id\"].nunique()}')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.616364Z","iopub.execute_input":"2024-02-10T01:12:27.616905Z","iopub.status.idle":"2024-02-10T01:12:27.876623Z","shell.execute_reply.started":"2024-02-10T01:12:27.616877Z","shell.execute_reply":"2024-02-10T01:12:27.875629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --------- Transform subsampels into sampels ---------\n#                     subsamples ===> samples\n\nVOTES = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n\n\ntrain = df.groupby('eeg_id').agg({\n    'spectrogram_id': 'first',\n    'patient_id': 'first',\n}).reset_index()\n\n# Adds together all the votes based on the EEG_ID\ntmp = df.groupby('eeg_id')[VOTES].agg('sum')\nfor v in VOTES:\n    train[v] = tmp[v].values\n\ntrain['target'] = train[VOTES].sum(axis=1)\ntrain[VOTES] = train[VOTES].div(train['target'], axis=0)\ntrain['target'] = train[VOTES].idxmax(axis=1)\n\n# Shape and View\nprint(f\"Train Shape: {train.shape}\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.881475Z","iopub.execute_input":"2024-02-10T01:12:27.881789Z","iopub.status.idle":"2024-02-10T01:12:27.946854Z","shell.execute_reply.started":"2024-02-10T01:12:27.881764Z","shell.execute_reply":"2024-02-10T01:12:27.945943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n#### DATASET","metadata":{}},{"cell_type":"code","source":"%%time\n\nREAD = False\n\n\n# NPY Data taken from Chris Deotte. Appreciate the smaller mem usage of .npy format \nif READ:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()\n    eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()\n    \n    example = train[\"spectrogram_id\"].iloc[0]\n    example_eeg = train[\"eeg_id\"].iloc[0]\n\n    print(f\"Size of EEG Dictionary: {len(all_eegs)}\")\n    print(f\"Shape of EEG Item: {all_eegs[example_eeg].shape}\")\n\n    print(f\"Size of Spectogram Dictionary: {len(spectrograms)}\")\n    print(f\"Shape of Spectogram Item: {spectrograms[example].shape}\")\nelse:\n    spectrograms = {}\n    eegs = {}\n    ","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.947912Z","iopub.execute_input":"2024-02-10T01:12:27.948171Z","iopub.status.idle":"2024-02-10T01:12:27.954744Z","shell.execute_reply.started":"2024-02-10T01:12:27.948148Z","shell.execute_reply":"2024-02-10T01:12:27.953903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Building a Data Generator\nMain Reference:\nhttps://stanford.edu/~shervine/blog/keras-how-to-generate-data-on-the-fly\n\nThings to do:\n   - Load from spectorgram & eeg dictionary\n   ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport keras\n\nclass DataGenerator(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self,\n                 data,\n                 state=\"train\", \n                 batch_size=32,\n                 n_classes=6,\n                 shuffle=True, \n                 specs=spectrograms,\n                 eegs=eegs):\n        'Initialization'\n        self.data = data\n        self.state = state\n        self.batch_size = batch_size\n        self.n_classes = n_classes\n        self.specs = specs\n        self.eegs = eegs\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.data) / self.batch_size))\n    \n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Generate data\n        X, y = self.__data_generation(indexes)\n\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.data))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        X = np.zeros((len(indexes),128,256,4),dtype='float32')\n        y = np.zeros((len(indexes),self.n_classes),dtype='float32')\n        img = np.ones((128,256),dtype='float32')\n        \n        for n, i in enumerate(indexes):\n            \n            # date row\n            cur = self.data.iloc[i]\n            \n            # images of spec and eeg\n            spec = self.specs[cur.spectrogram_id] # Not in use as of now\n            eeg = self.eegs[cur.eeg_id]\n\n            # add the data on\n            X[n] = eeg\n            \n            if self.state == \"train\":\n                y[n,] = cur[VOTES]\n        \n        return X, y\n        ","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.955916Z","iopub.execute_input":"2024-02-10T01:12:27.956222Z","iopub.status.idle":"2024-02-10T01:12:27.968925Z","shell.execute_reply.started":"2024-02-10T01:12:27.956200Z","shell.execute_reply":"2024-02-10T01:12:27.968224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the Data Shape and for Testing\nif READ:\n    gen = DataGenerator(train,\n                       batch_size=6)\n\n# gen.__getitem__(0)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.970301Z","iopub.execute_input":"2024-02-10T01:12:27.970595Z","iopub.status.idle":"2024-02-10T01:12:27.983307Z","shell.execute_reply.started":"2024-02-10T01:12:27.970572Z","shell.execute_reply":"2024-02-10T01:12:27.982457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model Definitions\n\nBuilding the Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Dense\n\n# building model\ndef build_cnn():\n    inp = tf.keras.Input(shape=(128, 256, 4))\n    \n    # EfficientNetB0 Base Model\n    base_model = tf.keras.applications.efficientnet.EfficientNetB0(include_top=False, weights=\"imagenet\", input_shape=None)\n\n    base_model.load_weights('/kaggle/input/keras-pretrained-models/EfficientNetB0_NoTop_ImageNet.h5')\n    \n    x = [inp[:,:,:,i:i+1] for i in range(4)]\n    x = tf.keras.layers.Concatenate(axis=1)(x)\n    \n    \n    # Model and Additional layera\n    x = base_model(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dense(6,activation='softmax', dtype='float32')(x)\n\n    model = tf.keras.Model(inputs=inp, outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate=1e-3)\n    loss = tf.keras.losses.KLDivergence()\n    \n    model.compile(loss=loss, optimizer=opt, metrics=['accuracy', 'kullback_leibler_divergence'])\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:27.984342Z","iopub.execute_input":"2024-02-10T01:12:27.984621Z","iopub.status.idle":"2024-02-10T01:12:28.109817Z","shell.execute_reply.started":"2024-02-10T01:12:27.984599Z","shell.execute_reply":"2024-02-10T01:12:28.108971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build \n\n* Using Multiple GPU's in order to Train and have enough RAM. This is defined in the strategy variable.\n* I have choosen to not use a validation data set though plan to in the future.\n* I also do not use the Spectrogram and only use EEG's for this model\n\n\n#### v0_HMS_ConvNet:\n\n534/534 [==============================] - 263s 371ms/step - loss: 0.7295 - accuracy: 0.6474 - kullback_leibler_divergence: 0.7295\nEpoch 2/4\n\n534/534 [==============================] - 198s 370ms/step - loss: 0.5857 - accuracy: 0.7108 - kullback_leibler_divergence: 0.5857\nEpoch 3/4\n\n534/534 [==============================] - 197s 369ms/step - loss: 0.5344 - accuracy: 0.7309 - kullback_leibler_divergence: 0.5344\nEpoch 4/4\n\n534/534 [==============================] - 197s 370ms/step - loss: 0.4868 - accuracy: 0.7488 - kullback_leibler_divergence: 0.4868","metadata":{}},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\n\nTRAIN = False\nSAVE = False\n\nif TRAIN:\n\n    gen =  DataGenerator(data=train, \n                        batch_size=32, \n                        n_classes=6,\n                        shuffle=True)\n\n    with strategy.scope():\n        v0_HMS_ConvNet = build_cnn() \n\n        v0_HMS_ConvNet._name = \"v0_HMS_ConvNet\"\n\n        v0_HMS_ConvNet.fit(gen,\n              verbose=1,  \n              epochs=4)\n        if SAVE:\n\n            filename = \"v0_HMS_ConvNet.pkl\"\n\n            with open(filename, 'wb') as file:  \n                pickle.dump(v0_HMS_ConvNet, file)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:28.110878Z","iopub.execute_input":"2024-02-10T01:12:28.111183Z","iopub.status.idle":"2024-02-10T01:12:29.204626Z","shell.execute_reply.started":"2024-02-10T01:12:28.111152Z","shell.execute_reply":"2024-02-10T01:12:29.203818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PREDICTIONS\n\nBegin with the correct formating of the data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\n\ntest_df = test_df.groupby('eeg_id').agg({\n    'spectrogram_id': 'first',\n    'patient_id': 'first',\n}).reset_index()\n\n\n# Shape and View\nprint(f\"Train Shape: {test_df.shape}\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:29.207319Z","iopub.execute_input":"2024-02-10T01:12:29.207639Z","iopub.status.idle":"2024-02-10T01:12:29.224601Z","shell.execute_reply.started":"2024-02-10T01:12:29.207612Z","shell.execute_reply":"2024-02-10T01:12:29.223709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Processing the Parquets\n\nCleaning the data is also taken from Chris Deotte\n- This includes his code to read the parquets\n","metadata":{}},{"cell_type":"code","source":"import pywt, librosa\n\nUSE_WAVELET = None \nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:29.226248Z","iopub.execute_input":"2024-02-10T01:12:29.226635Z","iopub.status.idle":"2024-02-10T01:12:29.325282Z","shell.execute_reply.started":"2024-02-10T01:12:29.226599Z","shell.execute_reply":"2024-02-10T01:12:29.324556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ ALL EEG SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 1\nEEG_IDS2 = test_df.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img\n    \n# READ ALL SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\nfiles2 = os.listdir(PATH2)\nprint(f'There are {len(files2)} test spectrogram parquets')\n    \nspectrograms2 = {}\nfor i,f in enumerate(files2):\n    if i%100==0: print(i,', ',end='')\n    tmp = pd.read_parquet(f'{PATH2}{f}')\n    name = int(f.split('.')[0])\n    spectrograms2[name] = tmp.iloc[:,1:].values","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:12:29.326346Z","iopub.execute_input":"2024-02-10T01:12:29.326969Z","iopub.status.idle":"2024-02-10T01:12:41.554380Z","shell.execute_reply.started":"2024-02-10T01:12:29.326943Z","shell.execute_reply":"2024-02-10T01:12:41.553281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = DataGenerator(test_df,\n                        state=\"test\",\n                        batch_size=64,\n                        shuffle=False,\n                        specs=spectrograms2,\n                        eegs=all_eegs2)\n\ntg = test_gen.__getitem__(0)\ntg = tg[0] # Getting the X (img)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:13:00.296789Z","iopub.execute_input":"2024-02-10T01:13:00.297452Z","iopub.status.idle":"2024-02-10T01:13:00.303205Z","shell.execute_reply.started":"2024-02-10T01:13:00.297411Z","shell.execute_reply":"2024-02-10T01:13:00.301950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LOAD = True\nMODEL_PATH = \"/kaggle/input/v0_hms_convnet_eeg_only.pkl/tensorflow1/hm-eeg-only/1/v0_HMS_ConvNet.pkl\"\n\nif LOAD:\n    with open(MODEL_PATH, 'rb') as f:\n        model = pickle.load(f)\n        \n        pred = model.predict(tg, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:13:03.264070Z","iopub.execute_input":"2024-02-10T01:13:03.265007Z","iopub.status.idle":"2024-02-10T01:13:43.155188Z","shell.execute_reply.started":"2024-02-10T01:13:03.264966Z","shell.execute_reply":"2024-02-10T01:13:43.154176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"est = pd.DataFrame({'eeg_id':test_df.eeg_id.values})\nest[VOTES] = pred\n\nest.to_csv('submission.csv',index=False)\ndisplay(est.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-10T01:13:43.157302Z","iopub.execute_input":"2024-02-10T01:13:43.157800Z","iopub.status.idle":"2024-02-10T01:13:43.178291Z","shell.execute_reply.started":"2024-02-10T01:13:43.157761Z","shell.execute_reply":"2024-02-10T01:13:43.177348Z"},"trusted":true},"execution_count":null,"outputs":[]}]}