{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7402356,"sourceType":"datasetVersion","datasetId":4304475},{"sourceId":7403069,"sourceType":"datasetVersion","datasetId":4304949},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":7450712,"sourceType":"datasetVersion","datasetId":4336944},{"sourceId":158958765,"sourceType":"kernelVersion"},{"sourceId":159333316,"sourceType":"kernelVersion"},{"sourceId":159396114,"sourceType":"kernelVersion"},{"sourceId":165868067,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notes","metadata":{}},{"cell_type":"code","source":"\"\"\"\nModel 1 is from Chris Deotte:\nhttps://www.kaggle.com/code/cdeotte/efficientnetb0-starter-lb-0-43\n\nModel 2 is from yunsuxiaozi\nhttps://www.kaggle.com/code/yunsuxiaozi/hms-baseline-resnet34d-512-512-inference-6-models\n\nModel(s) 3 is an ensemble of various gradient boosting models (XGBoost, CatBoost, LightGBM)\n\nThe strategy behind this notebook is a simple blended average of the \nbaseline resnet34d, the efficientnetB0 and multiple gradient boosting models. \n\nModel 1: 0.43\nModel 2: 0.45\nmodel 3: 0.58\n\nEnsemble: 0.39\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:19:47.447361Z","iopub.execute_input":"2024-03-04T20:19:47.448134Z","iopub.status.idle":"2024-03-04T20:19:47.463769Z","shell.execute_reply.started":"2024-03-04T20:19:47.448101Z","shell.execute_reply":"2024-03-04T20:19:47.462805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EfficientNet","metadata":{}},{"cell_type":"code","source":"# Needed imports\nimport os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport tensorflow as tf\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nprint('TensorFlow version =',tf.__version__)\n\n# Using multiple GPU's\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n\nVER = 5\n\n# IF THIS EQUALS NONE, THEN WE TRAIN NEW MODELS OTHERWISE WE LOAD PREVIOUSLY TRAINED MODELS\nLOAD_MODELS_FROM = '/kaggle/input/brain-efficientnet-models-v3-v4-v5/'\n\nUSE_KAGGLE_SPECTROGRAMS = True\nUSE_EEG_SPECTROGRAMS = True","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:19:47.465897Z","iopub.execute_input":"2024-03-04T20:19:47.466255Z","iopub.status.idle":"2024-03-04T20:20:06.687071Z","shell.execute_reply.started":"2024-03-04T20:19:47.466200Z","shell.execute_reply":"2024-03-04T20:20:06.686161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE MIXED PRECISION\nMIX = True\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:20:06.688470Z","iopub.execute_input":"2024-03-04T20:20:06.689469Z","iopub.status.idle":"2024-03-04T20:20:06.695097Z","shell.execute_reply.started":"2024-03-04T20:20:06.689432Z","shell.execute_reply":"2024-03-04T20:20:06.694103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get train data + the target column names\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:20:06.696368Z","iopub.execute_input":"2024-03-04T20:20:06.696694Z","iopub.status.idle":"2024-03-04T20:20:06.988113Z","shell.execute_reply.started":"2024-03-04T20:20:06.696664Z","shell.execute_reply":"2024-03-04T20:20:06.987237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create non-overlapping data\n\ntrain = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:20:06.991422Z","iopub.execute_input":"2024-03-04T20:20:06.991695Z","iopub.status.idle":"2024-03-04T20:20:07.100612Z","shell.execute_reply.started":"2024-03-04T20:20:06.991672Z","shell.execute_reply":"2024-03-04T20:20:07.099607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS (From Chris' dataset)\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:20:07.101829Z","iopub.execute_input":"2024-03-04T20:20:07.102179Z","iopub.status.idle":"2024-03-04T20:21:13.699657Z","shell.execute_reply.started":"2024-03-04T20:20:07.102147Z","shell.execute_reply":"2024-03-04T20:21:13.698682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_EEG_SPEC_FILES = False\n\n# Both from Chris again\nif READ_EEG_SPEC_FILES:\n    all_eegs = {}\n    for i,e in enumerate(train.eeg_id.values):\n        if i%100==0: print(i,', ',end='')\n        x = np.load(f'/kaggle/input/brain-eeg-spectrograms/EEG_Spectrograms/{e}.npy')\n        all_eegs[e] = x\nelse:\n    all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:21:13.700743Z","iopub.execute_input":"2024-03-04T20:21:13.701071Z","iopub.status.idle":"2024-03-04T20:22:36.806231Z","shell.execute_reply.started":"2024-03-04T20:21:13.701046Z","shell.execute_reply":"2024-03-04T20:22:36.805265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as albu\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\n# Here we efficiently generate batches of data fo training + data augmentation + preprocessing.\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, batch_size=32, shuffle=False, augment=False, mode='train',\n                 specs = spectrograms, eeg_specs = all_eegs): \n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.mode = mode\n        self.specs = specs\n        self.eeg_specs = eeg_specs\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int( np.ceil( len(self.data) / self.batch_size ) )\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X, y = self.__data_generation(indexes)\n        if self.augment: X = self.__augment_batch(X) \n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.data) )\n        if self.shuffle: np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' \n        \n        X = np.zeros((len(indexes),128,256,8),dtype='float32')\n        y = np.zeros((len(indexes),6),dtype='float32')\n        img = np.ones((128,256),dtype='float32')\n        \n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]\n            if self.mode=='test': \n                r = 0\n            else: \n                r = int( (row['min'] + row['max'])//4 )\n\n            for k in range(4):\n                # EXTRACT 300 ROWS OF SPECTROGRAM\n                img = self.specs[row.spec_id][r:r+300,k*100:(k+1)*100].T\n                \n                # LOG TRANSFORM SPECTROGRAM\n                img = np.clip(img,np.exp(-4),np.exp(8))\n                img = np.log(img)\n                \n                # STANDARDIZE PER IMAGE\n                ep = 1e-6\n                m = np.nanmean(img.flatten())\n                s = np.nanstd(img.flatten())\n                img = (img-m)/(s+ep)\n                img = np.nan_to_num(img, nan=0.0)\n                \n                # CROP TO 256 TIME STEPS\n                X[j,14:-14,:,k] = img[:,22:-22] / 2.0\n        \n            # EEG SPECTROGRAMS\n            img = self.eeg_specs[row.eeg_id]\n            X[j,:,:,4:] = img\n                \n            if self.mode!='test':\n                y[j,] = row[TARGETS]\n            \n        return X,y\n    \n    def __random_transform(self, img):\n        composition = albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            #albu.CoarseDropout(max_holes=8,max_height=32,max_width=32,fill_value=0,p=0.5),\n        ])\n        return composition(image=img)['image']\n            \n    def __augment_batch(self, img_batch):\n        for i in range(img_batch.shape[0]):\n            img_batch[i, ] = self.__random_transform(img_batch[i, ])\n        return img_batch","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:36.807511Z","iopub.execute_input":"2024-03-04T20:22:36.807811Z","iopub.status.idle":"2024-03-04T20:22:40.026500Z","shell.execute_reply.started":"2024-03-04T20:22:36.807785Z","shell.execute_reply":"2024-03-04T20:22:40.025545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot spectrograms\ngen = DataGenerator(train, batch_size=32, shuffle=False)\nROWS=2; COLS=3; BATCHES=2\n\nfor i,(x,y) in enumerate(gen):\n    plt.figure(figsize=(20,8))\n    for j in range(ROWS):\n        for k in range(COLS):\n            plt.subplot(ROWS,COLS,j*COLS+k+1)\n            t = y[j*COLS+k]\n            img = x[j*COLS+k,:,:,0][::-1,]\n            mn = img.flatten().min()\n            mx = img.flatten().max()\n            img = (img-mn)/(mx-mn)\n            plt.imshow(img)\n            tars = f'[{t[0]:0.2f}'\n            for s in t[1:]: tars += f', {s:0.2f}'\n            eeg = train.eeg_id.values[i*32+j*COLS+k]\n            plt.title(f'EEG = {eeg}\\nTarget = {tars}',size=12)\n            plt.yticks([])\n            plt.ylabel('Frequencies (Hz)',size=14)\n            plt.xlabel('Time (sec)',size=16)\n    plt.show()\n    if i==BATCHES-1: break","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:40.027801Z","iopub.execute_input":"2024-03-04T20:22:40.028432Z","iopub.status.idle":"2024-03-04T20:22:42.661025Z","shell.execute_reply.started":"2024-03-04T20:22:40.028403Z","shell.execute_reply":"2024-03-04T20:22:42.660139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make + plot training schedule (cosine) of the learning rate over the epochs.\n\nimport math\nLR_START = 1e-6\nLR_MAX = 1e-3\nLR_MIN = 1e-6\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS2 = 10\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS2 - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN\n    return lr\n\nrng = [i for i in range(EPOCHS2)]\nlr_y = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y, '-o')\nplt.xlabel('epoch',size=14); plt.ylabel('learning rate',size=14)\nplt.title('Cosine Training Schedule',size=16); plt.show()\n\nLR2 = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:42.662336Z","iopub.execute_input":"2024-03-04T20:22:42.662681Z","iopub.status.idle":"2024-03-04T20:22:42.883597Z","shell.execute_reply.started":"2024-03-04T20:22:42.662651Z","shell.execute_reply":"2024-03-04T20:22:42.882680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make + plot training schedule (step) of the learning rate over the epochs.\n\nLR_START = 1e-4\nLR_MAX = 1e-3\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 1\nLR_STEP_DECAY = 0.1\nEVERY = 1\nEPOCHS = 4\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = LR_MAX * LR_STEP_DECAY**((epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS)//EVERY)\n    return lr\n\nrng = [i for i in range(EPOCHS)]\ny = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, y, 'o-'); \nplt.xlabel('epoch',size=14); plt.ylabel('learning rate',size=14)\nplt.title('Step Training Schedule',size=16); plt.show()\n\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:42.884878Z","iopub.execute_input":"2024-03-04T20:22:42.885158Z","iopub.status.idle":"2024-03-04T20:22:43.104499Z","shell.execute_reply.started":"2024-03-04T20:22:42.885134Z","shell.execute_reply":"2024-03-04T20:22:43.103617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-index --find-links=/kaggle/input/tf-efficientnet-whl-files /kaggle/input/tf-efficientnet-whl-files/efficientnet-1.1.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:43.105750Z","iopub.execute_input":"2024-03-04T20:22:43.106103Z","iopub.status.idle":"2024-03-04T20:22:58.092282Z","shell.execute_reply.started":"2024-03-04T20:22:43.106072Z","shell.execute_reply":"2024-03-04T20:22:58.091140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build EfficientNet\n\nimport efficientnet.tfkeras as efn\n\ndef build_model():\n    \n    inp = tf.keras.Input(shape=(128,256,8))\n    base_model = efn.EfficientNetB0(include_top=False, weights=None, input_shape=None)\n    base_model.load_weights('/kaggle/input/tf-efficientnet-imagenet-weights/efficientnet-b0_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5')\n    \n    # RESHAPE INPUT 128x256x8 => 512x512x3 MONOTONE IMAGE\n    # KAGGLE SPECTROGRAMS\n    x1 = [inp[:,:,:,i:i+1] for i in range(4)]\n    x1 = tf.keras.layers.Concatenate(axis=1)(x1)\n    # EEG SPECTROGRAMS\n    x2 = [inp[:,:,:,i+4:i+5] for i in range(4)]\n    x2 = tf.keras.layers.Concatenate(axis=1)(x2)\n    # MAKE 512X512X3\n    if USE_KAGGLE_SPECTROGRAMS & USE_EEG_SPECTROGRAMS:\n        x = tf.keras.layers.Concatenate(axis=2)([x1,x2])\n    elif USE_EEG_SPECTROGRAMS: x = x2\n    else: x = x1\n    x = tf.keras.layers.Concatenate(axis=3)([x,x,x])\n    \n    # OUTPUT\n    x = base_model(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dense(6,activation='softmax', dtype='float32')(x)\n        \n    # COMPILE MODEL\n    model = tf.keras.Model(inputs=inp, outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate = 1e-3)\n    loss = tf.keras.losses.KLDivergence()\n\n    model.compile(loss=loss, optimizer = opt) \n        \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:58.093908Z","iopub.execute_input":"2024-03-04T20:22:58.094242Z","iopub.status.idle":"2024-03-04T20:22:58.118490Z","shell.execute_reply.started":"2024-03-04T20:22:58.094204Z","shell.execute_reply":"2024-03-04T20:22:58.117642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train EfficientNet\n\nfrom sklearn.model_selection import KFold, GroupKFold\nimport tensorflow.keras.backend as K, gc\n\nall_oof = []\nall_true = []\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):  \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    \n    train_gen = DataGenerator(train.iloc[train_index], shuffle=True, batch_size=32, augment=False)\n    valid_gen = DataGenerator(train.iloc[valid_index], shuffle=False, batch_size=64, mode='valid')\n    \n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    K.clear_session()\n    with strategy.scope():\n        model = build_model()\n    if LOAD_MODELS_FROM is None:\n        model.fit(train_gen, verbose=1,\n              validation_data = valid_gen,\n              epochs=EPOCHS, callbacks = [LR])\n        model.save_weights(f'EffNet_v{VER}_f{i}.h5')\n    else:\n        model.load_weights(f'{LOAD_MODELS_FROM}EffNet_v{VER}_f{i}.h5')\n        \n    oof = model.predict(valid_gen, verbose=1)\n    all_oof.append(oof)\n    all_true.append(train.iloc[valid_index][TARGETS].values)\n    \n    del model, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:22:58.124433Z","iopub.execute_input":"2024-03-04T20:22:58.124715Z","iopub.status.idle":"2024-03-04T20:25:56.328073Z","shell.execute_reply.started":"2024-03-04T20:22:58.124691Z","shell.execute_reply":"2024-03-04T20:25:56.327266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/kaggle-kl-div')\nfrom kaggle_kl_div import score\n\noof = pd.DataFrame(all_oof.copy())\noof['id'] = np.arange(len(oof))\n\ntrue = pd.DataFrame(all_true.copy())\ntrue['id'] = np.arange(len(true))\n\ncv = score(solution=true, submission=oof, row_id_column_name='id')\nprint('CV Score KL-Div for EfficientNetB2 =',cv)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:25:56.329935Z","iopub.execute_input":"2024-03-04T20:25:56.330206Z","iopub.status.idle":"2024-03-04T20:25:56.409135Z","shell.execute_reply.started":"2024-03-04T20:25:56.330184Z","shell.execute_reply":"2024-03-04T20:25:56.408267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we use it for test dataset\ndel all_eegs, spectrograms; gc.collect()\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:25:56.410322Z","iopub.execute_input":"2024-03-04T20:25:56.410603Z","iopub.status.idle":"2024-03-04T20:25:56.626833Z","shell.execute_reply.started":"2024-03-04T20:25:56.410580Z","shell.execute_reply":"2024-03-04T20:25:56.625916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\nfiles2 = os.listdir(PATH2)\nprint(f'There are {len(files2)} test spectrogram parquets')\n    \nspectrograms2 = {}\nfor i,f in enumerate(files2):\n    if i%100==0: print(i,', ',end='')\n    tmp = pd.read_parquet(f'{PATH2}{f}')\n    name = int(f.split('.')[0])\n    spectrograms2[name] = tmp.iloc[:,1:].values\n    \n# RENAME FOR DATALOADER\ntest = test.rename({'spectrogram_id':'spec_id'},axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:25:56.627849Z","iopub.execute_input":"2024-03-04T20:25:56.628123Z","iopub.status.idle":"2024-03-04T20:25:57.073166Z","shell.execute_reply.started":"2024-03-04T20:25:56.628099Z","shell.execute_reply":"2024-03-04T20:25:57.072404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# spectrogram from eeg again, same idea as what Chris Deotte did\n\nimport pywt, librosa\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:25:57.074679Z","iopub.execute_input":"2024-03-04T20:25:57.075048Z","iopub.status.idle":"2024-03-04T20:25:57.103282Z","shell.execute_reply.started":"2024-03-04T20:25:57.075014Z","shell.execute_reply":"2024-03-04T20:25:57.102388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 1\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET BY USING THE EARLIER DEFINED FUNCTION\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:25:57.105883Z","iopub.execute_input":"2024-03-04T20:25:57.106181Z","iopub.status.idle":"2024-03-04T20:26:09.723064Z","shell.execute_reply.started":"2024-03-04T20:25:57.106157Z","shell.execute_reply":"2024-03-04T20:26:09.722056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER EFFICIENTNET ON TEST + SAVE THE SUBMISSION ONLY FOR THIS MODEL; WILL BE COMBINED LATER ON\npreds = []\nmodel = build_model()\ntest_gen = DataGenerator(test, shuffle=False, batch_size=64, mode='test',\n                         specs = spectrograms2, eeg_specs = all_eegs2)\n\nfor i in range(5):\n    print(f'Fold {i+1}')\n    if LOAD_MODELS_FROM:\n        model.load_weights(f'{LOAD_MODELS_FROM}EffNet_v{VER}_f{i}.h5')\n    else:\n        model.load_weights(f'EffNet_v{VER}_f{i}.h5')\n    pred = model.predict(test_gen, verbose=1)\n    preds.append(pred)\npred = np.mean(preds,axis=0)\nprint()\nprint('Test preds shape',pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:09.724493Z","iopub.execute_input":"2024-03-04T20:26:09.725614Z","iopub.status.idle":"2024-03-04T20:26:17.578519Z","shell.execute_reply.started":"2024-03-04T20:26:09.725575Z","shell.execute_reply":"2024-03-04T20:26:17.577544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MAKE CORRECT SUBMISSION SHAPE\nsub1 = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub1[TARGETS] = pred\nprint('Submission shape',sub1.shape)\nsub1.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:17.579548Z","iopub.execute_input":"2024-03-04T20:26:17.579833Z","iopub.status.idle":"2024-03-04T20:26:17.595167Z","shell.execute_reply.started":"2024-03-04T20:26:17.579806Z","shell.execute_reply":"2024-03-04T20:26:17.594170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_combine = pred\npreds_combine","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:17.596422Z","iopub.execute_input":"2024-03-04T20:26:17.596696Z","iopub.status.idle":"2024-03-04T20:26:17.608287Z","shell.execute_reply.started":"2024-03-04T20:26:17.596673Z","shell.execute_reply":"2024-03-04T20:26:17.607392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub1.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:17.609521Z","iopub.execute_input":"2024-03-04T20:26:17.609875Z","iopub.status.idle":"2024-03-04T20:26:17.623930Z","shell.execute_reply.started":"2024-03-04T20:26:17.609844Z","shell.execute_reply":"2024-03-04T20:26:17.622957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ResNet","metadata":{}},{"cell_type":"code","source":"# Needed imports\n#https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-training\n#https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-inference\nimport pandas as pd\nimport numpy as np\nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.transforms as transforms\nimport random\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:17.625043Z","iopub.execute_input":"2024-03-04T20:26:17.625381Z","iopub.status.idle":"2024-03-04T20:26:23.562524Z","shell.execute_reply.started":"2024-03-04T20:26:17.625346Z","shell.execute_reply":"2024-03-04T20:26:23.561546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    seed=2024\n    image_transform=transforms.Resize((512, 512))\n    num_folds=5","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:23.563702Z","iopub.execute_input":"2024-03-04T20:26:23.564493Z","iopub.status.idle":"2024-03-04T20:26:23.569424Z","shell.execute_reply.started":"2024-03-04T20:26:23.564449Z","shell.execute_reply":"2024-03-04T20:26:23.568500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading the models used\nmodels=[]\nfor i in range(Config.num_folds):\n    model = torch.load(f'/kaggle/input/hms-baseline-resnet34d-512-512-training-5-folds/HMS_resnet_fold{i}.pth')\n    models.append(model)\nmodel = torch.load(\"/kaggle/input/hms-baseline-resnet34d-512-512-training/HMS_resnet.pth\")\nmodels.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:23.570458Z","iopub.execute_input":"2024-03-04T20:26:23.570734Z","iopub.status.idle":"2024-03-04T20:26:31.968149Z","shell.execute_reply.started":"2024-03-04T20:26:23.570710Z","shell.execute_reply":"2024-03-04T20:26:31.967291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n    torch.manual_seed(seed)\n    np.random.seed(seed)\n    random.seed(seed)\nseed_everything(Config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:31.969399Z","iopub.execute_input":"2024-03-04T20:26:31.969689Z","iopub.status.idle":"2024-03-04T20:26:31.984462Z","shell.execute_reply.started":"2024-03-04T20:26:31.969664Z","shell.execute_reply":"2024-03-04T20:26:31.983517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load necessary data\ntest_df=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nsubmission=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nsubmission=submission.merge(test_df,on='eeg_id',how='left')\nsubmission['path']=submission['spectrogram_id'].apply(lambda x: \"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/\"+str(x)+\".parquet\" )\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:31.985657Z","iopub.execute_input":"2024-03-04T20:26:31.985989Z","iopub.status.idle":"2024-03-04T20:26:32.025137Z","shell.execute_reply.started":"2024-03-04T20:26:31.985947Z","shell.execute_reply":"2024-03-04T20:26:32.023929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions on test\npaths=submission['path'].values\ntest_preds=[]\nfor path in paths:\n    eps=1e-6\n    data=pd.read_parquet(path)\n    data = data.fillna(-1).values[:,1:].T\n    data=data[:,0:300]#(400,300)\n    data=np.clip(data,np.exp(-6),np.exp(10))\n    data= np.log(data)\n    data_mean=data.mean(axis=(0,1))\n    data_std=data.std(axis=(0,1))\n    data=(data-data_mean)/(data_std+eps)\n    data_tensor = torch.unsqueeze(torch.Tensor(data), dim=0)\n    data=Config.image_transform(data_tensor)\n    test_pred=[]\n    for model in models:\n        model.eval()\n        with torch.no_grad():\n            pred=F.softmax(model(data.unsqueeze(0)))[0]\n            pred=pred.detach().cpu().numpy()\n        test_pred.append(pred)\n    test_pred=np.array(test_pred).mean(axis=0)\n    test_preds.append(test_pred)\ntest_preds=np.array(test_preds)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:32.026489Z","iopub.execute_input":"2024-03-04T20:26:32.026825Z","iopub.status.idle":"2024-03-04T20:26:34.040614Z","shell.execute_reply.started":"2024-03-04T20:26:32.026796Z","shell.execute_reply":"2024-03-04T20:26:34.039612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission 2\nsub2=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nlabels=['seizure','lpd','gpd','lrda','grda','other']\nfor i in range(len(labels)):\n    sub2[f'{labels[i]}_vote']=test_preds[:,i]\nsub2.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.042166Z","iopub.execute_input":"2024-03-04T20:26:34.042567Z","iopub.status.idle":"2024-03-04T20:26:34.063231Z","shell.execute_reply.started":"2024-03-04T20:26:34.042534Z","shell.execute_reply":"2024-03-04T20:26:34.062245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"preds_combine","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.065891Z","iopub.execute_input":"2024-03-04T20:26:34.066398Z","iopub.status.idle":"2024-03-04T20:26:34.073556Z","shell.execute_reply.started":"2024-03-04T20:26:34.066365Z","shell.execute_reply":"2024-03-04T20:26:34.072545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final submission for the Deep Learning part of the ensemble\nsubmission_DL=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nlabels=['seizure','lpd','gpd','lrda','grda','other']\nfor i in range(len(labels)):\n    submission_DL[f'{labels[i]}_vote']=(test_preds[:,i] + preds_combine[:, i])/2\n# submission_DL.to_csv(\"submission.csv\",index=None)\ndisplay(submission_DL.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.075277Z","iopub.execute_input":"2024-03-04T20:26:34.075553Z","iopub.status.idle":"2024-03-04T20:26:34.100048Z","shell.execute_reply.started":"2024-03-04T20:26:34.075523Z","shell.execute_reply":"2024-03-04T20:26:34.099201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsubmission_DL.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.101157Z","iopub.execute_input":"2024-03-04T20:26:34.101444Z","iopub.status.idle":"2024-03-04T20:26:34.109532Z","shell.execute_reply.started":"2024-03-04T20:26:34.101421Z","shell.execute_reply":"2024-03-04T20:26:34.108565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gradient boosting models\n\n### Credits to York Yong for sharing his notebook so we can use it as our baseline (https://www.kaggle.com/code/yorkyong/exploring-eeg-a-beginner-s-guide)\n\n### Also thanks to Chris Deotte for his work on transforming eeg data to spectrograms.","metadata":{}},{"cell_type":"code","source":"# Needed imports\nimport numpy as np\nimport pandas as pd\nimport os\nprint(os.listdir(\"/kaggle/input/gradboosttrainnew\"))\nimport warnings\nwarnings.filterwarnings('ignore')\nimport pywt, librosa\nfrom sklearn.impute import SimpleImputer\nimport catboost as cat\nfrom catboost import CatBoostClassifier, Pool\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold, GroupKFold\nimport xgboost as xgb\nVersion = 1","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.111521Z","iopub.execute_input":"2024-03-04T20:26:34.111864Z","iopub.status.idle":"2024-03-04T20:26:34.130900Z","shell.execute_reply.started":"2024-03-04T20:26:34.111838Z","shell.execute_reply":"2024-03-04T20:26:34.129959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check test data\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.132004Z","iopub.execute_input":"2024-03-04T20:26:34.141415Z","iopub.status.idle":"2024-03-04T20:26:34.155599Z","shell.execute_reply.started":"2024-03-04T20:26:34.141383Z","shell.execute_reply":"2024-03-04T20:26:34.154714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features created + define functions to get spectrograms from eeg data\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\n\n\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mdn_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_mdn_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_20s' for c in SPEC_COLS]\n\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\n\n\n# FROM CHRIS DEOTTE\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.157094Z","iopub.execute_input":"2024-03-04T20:26:34.157873Z","iopub.status.idle":"2024-03-04T20:26:34.242322Z","shell.execute_reply.started":"2024-03-04T20:26:34.157841Z","shell.execute_reply":"2024-03-04T20:26:34.241416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating spectrogram from eeg data\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 0\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.243478Z","iopub.execute_input":"2024-03-04T20:26:34.243777Z","iopub.status.idle":"2024-03-04T20:26:34.622399Z","shell.execute_reply.started":"2024-03-04T20:26:34.243748Z","shell.execute_reply":"2024-03-04T20:26:34.621258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n\nfor k in range(len(test)):\n    row = test.iloc[k]\n    r = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{r}.parquet')\n\n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean(spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin(spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n    x = np.nanmax(spec.iloc[:,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanstd(spec.iloc[:,1:].values, axis=0)\n    data[k,1200:1600] = x\n    x = np.nanmedian(spec.iloc[:,1:].values, axis=0)\n    data[k,1600:2000] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 25, axis=0)\n    data[k,2000:2400] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 50, axis=0)\n    data[k,2400:2800] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 75, axis=0)\n    data[k,2800:3200] = x\n    \n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean(spec.iloc[145:155,1:].values, axis=0)\n    data[k,3200:3600] = x\n    x = np.nanmin(spec.iloc[145:155,1:].values, axis=0)\n    data[k,3600:4000] = x\n    x = np.nanmax(spec.iloc[145:155,1:].values, axis=0)\n    data[k,4000:4400] = x\n    x = np.nanstd(spec.iloc[145:155,1:].values, axis=0)\n    data[k,4400:4800] = x\n    x = np.nanmedian(spec.iloc[145:155,1:].values, axis=0)\n    data[k,4800:5200] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 25, axis=0)\n    data[k,5200:5600] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 50, axis=0)\n    data[k,5600:6000] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 75, axis=0)\n    data[k,6000:6400] = x\n\n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs2[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6400:6912] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6912:7424] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,7424:7936] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,7936:8448] = x\n\ntest[FEATURES] = data\nprint(); print('New test shape:',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:34.624299Z","iopub.execute_input":"2024-03-04T20:26:34.624975Z","iopub.status.idle":"2024-03-04T20:26:48.699089Z","shell.execute_reply.started":"2024-03-04T20:26:34.624935Z","shell.execute_reply":"2024-03-04T20:26:48.698115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace NaN with median (NOT mean due to it being more vulnerable to outliers)\n\n# test.fillna(test.mean(), inplace=True)\n\nnan_per_column = test.isnull().sum().sum()\nprint(nan_per_column)\n\nimputer = SimpleImputer(strategy='median')\n\nnumeric_columns = test.select_dtypes(include=['number']).columns\n\nimputer.fit(test[numeric_columns])\n\ntest_imputed = imputer.transform(test[numeric_columns])\n\ntest[numeric_columns] = pd.DataFrame(test_imputed, columns=numeric_columns)\n\nnan_per_column = test.isnull().sum().sum()\nprint(nan_per_column)\n\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:48.700090Z","iopub.execute_input":"2024-03-04T20:26:48.700363Z","iopub.status.idle":"2024-03-04T20:26:53.595834Z","shell.execute_reply.started":"2024-03-04T20:26:48.700340Z","shell.execute_reply":"2024-03-04T20:26:53.594882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER LightGBM ON TEST\npreds_lgbm = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    model = lgb.Booster(model_file=f'/kaggle/input/gradboosttrainnew/LightGBM_v{Version}_f{i}.txt')\n    \n    pred_lgbm = model.predict(test[FEATURES])\n    preds_lgbm.append(pred_lgbm)\n\n# Average the predictions from each fold\npred_lgbm = np.mean(preds_lgbm, axis=0)\nprint()\nprint('Test preds shape', pred_lgbm.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:26:53.597055Z","iopub.execute_input":"2024-03-04T20:26:53.597386Z","iopub.status.idle":"2024-03-04T20:27:01.284271Z","shell.execute_reply.started":"2024-03-04T20:26:53.597359Z","shell.execute_reply":"2024-03-04T20:27:01.283271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER XGBoost ON TEST\npreds_xgb = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    model = xgb.XGBClassifier()\n    model.load_model(f'/kaggle/input/gradboosttrainnew/XGB_v{Version}_f{i}.model')\n    \n    pred_xgb = model.predict_proba(test[FEATURES])\n    preds_xgb.append(pred_xgb)\n\n# Average the predictions from each fold\npred_xgb = np.mean(preds_xgb, axis=0)\nprint()\nprint('Test preds shape', pred_xgb.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:01.291000Z","iopub.execute_input":"2024-03-04T20:27:01.291318Z","iopub.status.idle":"2024-03-04T20:27:08.869923Z","shell.execute_reply.started":"2024-03-04T20:27:01.291291Z","shell.execute_reply":"2024-03-04T20:27:08.868696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER CatBoost ON TEST\npreds_cat = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    model.load_model(f'/kaggle/input/gradboosttrainnew/CAT_v{Version}_f{i}.cat')\n    \n    pred_cat = model.predict_proba(test[FEATURES])\n    preds_cat.append(pred_cat)\n\n# Average the predictions from each fold\npred_cat = np.mean(preds_cat, axis=0)\nprint()\nprint('Test preds shape', pred_cat.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:08.871149Z","iopub.execute_input":"2024-03-04T20:27:08.871521Z","iopub.status.idle":"2024-03-04T20:27:14.206555Z","shell.execute_reply.started":"2024-03-04T20:27:08.871492Z","shell.execute_reply":"2024-03-04T20:27:14.205530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take the mean of the gradient boosting models predictions\n\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\nTARGETS = df.columns[-6:]\n\nsub = pd.DataFrame({'eeg_id':test.eeg_id.values})\n    \npred = np.mean([pred_cat, pred_xgb, pred_lgbm], axis=0)\n\nsub[TARGETS] = pred\n\n\n#sub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.208076Z","iopub.execute_input":"2024-03-04T20:27:14.208524Z","iopub.status.idle":"2024-03-04T20:27:14.547179Z","shell.execute_reply.started":"2024-03-04T20:27:14.208486Z","shell.execute_reply":"2024-03-04T20:27:14.546196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_DL.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.548378Z","iopub.execute_input":"2024-03-04T20:27:14.548682Z","iopub.status.idle":"2024-03-04T20:27:14.559234Z","shell.execute_reply.started":"2024-03-04T20:27:14.548656Z","shell.execute_reply":"2024-03-04T20:27:14.558358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.560419Z","iopub.execute_input":"2024-03-04T20:27:14.560764Z","iopub.status.idle":"2024-03-04T20:27:14.574566Z","shell.execute_reply.started":"2024-03-04T20:27:14.560733Z","shell.execute_reply":"2024-03-04T20:27:14.573571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final submission being the combination of the deep learning part with a weight of 0.9 and\n# the gradient boosting part with a weight of 0.1, more info on this can be found in the report.\n\nweight_DL = 0.9\n\nweight_GB = 0.1\n\nmerged_sub = pd.merge(sub, submission_DL, on='eeg_id')\n\nfor col in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    merged_sub[col] = weight_GB * merged_sub[col + '_x'] + weight_DL * merged_sub[col + '_y']\n\nmerged_sub.drop(columns=['seizure_vote_x', 'lpd_vote_x', 'gpd_vote_x', 'lrda_vote_x', 'grda_vote_x', 'other_vote_x', 'seizure_vote_y', 'lpd_vote_y', 'gpd_vote_y', 'lrda_vote_y', 'grda_vote_y', 'other_vote_y'], inplace=True)\n\nmerged_sub\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.575909Z","iopub.execute_input":"2024-03-04T20:27:14.576206Z","iopub.status.idle":"2024-03-04T20:27:14.604569Z","shell.execute_reply.started":"2024-03-04T20:27:14.576180Z","shell.execute_reply":"2024-03-04T20:27:14.603763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_sub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.605641Z","iopub.execute_input":"2024-03-04T20:27:14.605926Z","iopub.status.idle":"2024-03-04T20:27:14.622794Z","shell.execute_reply.started":"2024-03-04T20:27:14.605902Z","shell.execute_reply":"2024-03-04T20:27:14.621966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission\nmerged_sub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T20:27:14.624061Z","iopub.execute_input":"2024-03-04T20:27:14.624402Z","iopub.status.idle":"2024-03-04T20:27:14.639111Z","shell.execute_reply.started":"2024-03-04T20:27:14.624371Z","shell.execute_reply":"2024-03-04T20:27:14.638259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}