{"metadata":{"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7402356,"sourceType":"datasetVersion","datasetId":4304475},{"sourceId":7403069,"sourceType":"datasetVersion","datasetId":4304949},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":7450712,"sourceType":"datasetVersion","datasetId":4336944},{"sourceId":7955448,"sourceType":"datasetVersion","datasetId":4679081},{"sourceId":158958765,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"papermill":{"default_parameters":{},"duration":270.012179,"end_time":"2024-01-14T22:56:02.916427","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-01-14T22:51:32.904248","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc\n# os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport tensorflow as tf\nimport pandas as pd, numpy as np\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.fft as fft\nimport scipy.signal as signal\nprint('TensorFlow version =',tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:40.323788Z","iopub.execute_input":"2024-03-27T10:58:40.324516Z","iopub.status.idle":"2024-03-27T10:58:53.268665Z","shell.execute_reply.started":"2024-03-27T10:58:40.324478Z","shell.execute_reply":"2024-03-27T10:58:53.267677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"this notebook originated from chris's notebook. thank you","metadata":{}},{"cell_type":"code","source":"tf.config.list_physical_devices('GPU')","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:53.270666Z","iopub.execute_input":"2024-03-27T10:58:53.271606Z","iopub.status.idle":"2024-03-27T10:58:53.771240Z","shell.execute_reply.started":"2024-03-27T10:58:53.271569Z","shell.execute_reply":"2024-03-27T10:58:53.770079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = '/kaggle'","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:53.772665Z","iopub.execute_input":"2024-03-27T10:58:53.773029Z","iopub.status.idle":"2024-03-27T10:58:53.794071Z","shell.execute_reply.started":"2024-03-27T10:58:53.772996Z","shell.execute_reply":"2024-03-27T10:58:53.793304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE MULTIPLE GPUS\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1:\n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse:\n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n\nVER = 5\n\n# IF THIS EQUALS NONE, THEN WE TRAIN NEW MODELS\n# IF THIS EQUALS DISK PATH, THEN WE LOAD PREVIOUSLY TRAINED MODELS\n# LOAD_MODELS_FROM = os.path.join(root_path, 'input','brain-efficientnet-models-v3-v4-v5')\nLOAD_MODELS_FROM = None\nUSE_KAGGLE_SPECTROGRAMS = True\nUSE_EEG_SPECTROGRAMS = True\n\n# USE_WAVELET = None #or \"db8\" or anything below\nUSE_WAVELET_RAW_EEG = \"db8\" #or \"db8\" or anything below\nUSE_WAVELET = None","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:53.796144Z","iopub.execute_input":"2024-03-27T10:58:53.796427Z","iopub.status.idle":"2024-03-27T10:58:54.670080Z","shell.execute_reply.started":"2024-03-27T10:58:53.796399Z","shell.execute_reply":"2024-03-27T10:58:54.669096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE MIXED PRECISION\nMIX = True\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:54.671507Z","iopub.execute_input":"2024-03-27T10:58:54.671932Z","iopub.status.idle":"2024-03-27T10:58:54.678040Z","shell.execute_reply.started":"2024-03-27T10:58:54.671895Z","shell.execute_reply":"2024-03-27T10:58:54.677147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as albu\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\ndef get_preprocessed_image_train(spec, eeg_spec, min, max):\n    \n    X = np.ones((128,256,8),dtype='float32')\n    r = int( (min + max)//4 )\n\n    for k in range(4):\n        # EXTRACT 300 ROWS OF SPECTROGRAM\n        img = spec[r:r+300,k*100:(k+1)*100].T\n\n        # LOG TRANSFORM SPECTROGRAM\n        img = np.clip(img,np.exp(-4),np.exp(8))\n        img = np.log(img)\n\n        # STANDARDIZE PER IMAGE\n        ep = 1e-6\n        m = np.nanmean(img.flatten())\n        s = np.nanstd(img.flatten())\n        img = (img-m)/(s+ep)\n        img = np.nan_to_num(img, nan=0.0)\n\n        # CROP TO 256 TIME STEPS\n        X[14:-14,:,k] = img[:,22:-22] / 2.0\n\n    # EEG SPECTROGRAMS\n    img = eeg_spec\n    X[:,:,4:] = img\n\n    return X\n\ndef get_preprocessed_image_test(spec, eeg_spec):\n    \n    X = np.ones((128,256,8),dtype='float32')\n    r = 0\n\n    for k in range(4):\n        # EXTRACT 300 ROWS OF SPECTROGRAM\n        img = spec[r:r+300,k*100:(k+1)*100].T\n\n        # LOG TRANSFORM SPECTROGRAM\n        img = np.clip(img,np.exp(-4),np.exp(8))\n        img = np.log(img)\n\n        # STANDARDIZE PER IMAGE\n        ep = 1e-6\n        m = np.nanmean(img.flatten())\n        s = np.nanstd(img.flatten())\n        img = (img-m)/(s+ep)\n        img = np.nan_to_num(img, nan=0.0)\n\n        # CROP TO 256 TIME STEPS\n        X[14:-14,:,k] = img[:,22:-22] / 2.0\n\n    # EEG SPECTROGRAMS\n    img = eeg_spec\n    X[:,:,4:] = img\n\n    return X\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:54.679532Z","iopub.execute_input":"2024-03-27T10:58:54.679856Z","iopub.status.idle":"2024-03-27T10:58:56.082545Z","shell.execute_reply.started":"2024-03-27T10:58:54.679826Z","shell.execute_reply":"2024-03-27T10:58:56.081311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* data generator","metadata":{}},{"cell_type":"code","source":"import albumentations as albu\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data,specs,eeg_specs,raw_eegs, signal_scaler, batch_size=32, shuffle=False, augment=False, mode='train',):\n\n        self.data = data\n        self.specs = specs\n        self.eeg_specs = eeg_specs\n        self.raw_eegs = raw_eegs\n        self.signal_scaler = signal_scaler\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.mode = mode\n        \n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int( np.ceil( len(self.data) / self.batch_size ) )\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X, y = self.__data_generation(indexes)\n        # if self.augment: X = self.__augment_batch(X)\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.data) )\n        if self.shuffle: np.random.shuffle(self.indexes)\n\n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples'\n\n        # X_2d = np.zeros((len(indexes),128,256,8),dtype='float32')\n        # X_1d = np.zeros((len(indexes),10000,4),dtype='float32')\n        # target_y = np.zeros((len(indexes),6),dtype='float32')\n        \n        # for j,i in enumerate(indexes):\n\n        #     X_2d[j,:,:,:] = self.x_2d[i,:,:,:]\n        #     X_1d[j,:,:] = self.x_1d[i,:,:]\n            \n        #     if self.mode != \"test\":\n        #         target_y[j] = self.target[i]\n\n        # return [X_2d, X_1d], target_y\n        X = np.zeros((len(indexes),128,256,8),dtype='float32')\n        y = np.zeros((len(indexes),6),dtype='float32')\n\n        x_middle_signal = np.zeros((len(indexes),10000,4),dtype='float32')\n\n        # x_middle_signal = all_middle_raw_eegs['middle_signal'][all_middle_raw_eegs['eeg_id'] == eeg_id]\n        img = np.ones((128,256),dtype='float32')\n\n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]\n            if self.mode=='test':\n                r = 0\n            else:\n                r = int( (row['min'] + row['max'])//4 )\n\n            for k in range(4):\n                # EXTRACT 300 ROWS OF SPECTROGRAM\n                img = self.specs[row.spec_id][r:r+300,k*100:(k+1)*100].T\n\n                # LOG TRANSFORM SPECTROGRAM\n                img = np.clip(img,np.exp(-4),np.exp(8))\n                img = np.log(img)\n\n                # STANDARDIZE PER IMAGE\n                ep = 1e-6\n                m = np.nanmean(img.flatten())\n                s = np.nanstd(img.flatten())\n                img = (img-m)/(s+ep)\n                img = np.nan_to_num(img, nan=0.0)\n\n                # CROP TO 256 TIME STEPS\n                X[j,14:-14,:,k] = img[:,22:-22] / 2.0\n\n            # EEG SPECTROGRAMS\n            img = self.eeg_specs[row.eeg_id]\n            X[j,:,:,4:] = img\n\n            # raw eeg\n            x_middle_signal[j,:,:] = self.raw_eegs['middle_signal'][self.raw_eegs['eeg_id'] == row.eeg_id]\n\n            if self.mode!='test':\n                y[j,] = row[TARGETS]\n        \n        for i in range(4):\n            x_middle_signal[:,:,i:i+1] = (x_middle_signal[:,:,i:i+1] - self.signal_scaler[f\"{i}\"][0])/(self.signal_scaler[f\"{i}\"][1]-self.signal_scaler[f\"{i}\"][0])\n\n        return [X,x_middle_signal],y\n\n    def __random_transform(self, img):\n        composition = albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            #albu.CoarseDropout(max_holes=8,max_height=32,max_width=32,fill_value=0,p=0.5),\n        ])\n        return composition(image=img)['image']\n\n    # def __augment_batch(self, img_batch):\n    #     for i in range(img_batch.shape[0]):\n    #         img_batch[i, ] = self.__random_transform(img_batch[i, ])\n    #     return img_batch","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:56.084306Z","iopub.execute_input":"2024-03-27T10:58:56.085031Z","iopub.status.idle":"2024-03-27T10:58:56.229293Z","shell.execute_reply.started":"2024-03-27T10:58:56.084995Z","shell.execute_reply":"2024-03-27T10:58:56.228473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### make trainset","metadata":{}},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nPATH = os.path.join(root_path, 'input','hms-harmful-brain-activity-classification', 'train_spectrograms')\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:\n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load(os.path.join(root_path, 'input','brain-spectrograms', 'specs.npy'),allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:58:56.230720Z","iopub.execute_input":"2024-03-27T10:58:56.231455Z","iopub.status.idle":"2024-03-27T10:59:59.040281Z","shell.execute_reply.started":"2024-03-27T10:58:56.231419Z","shell.execute_reply":"2024-03-27T10:59:59.039299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_EEG_SPEC_FILES = False\n\nif READ_EEG_SPEC_FILES:\n    all_eegs = {}\n    for i,e in enumerate(train.eeg_id.values):\n        if i%100==0: print(i,', ',end='')\n        x = np.load(os.path.join(root_path, 'input','brain-eeg-spectrograms', 'EEG_Spectrograms',f'{e}.npy'))\n        all_eegs[e] = x\nelse:\n    all_eegs = np.load(os.path.join(root_path, 'input','brain-eeg-spectrograms', 'eeg_specs.npy'),allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T10:59:59.041704Z","iopub.execute_input":"2024-03-27T10:59:59.042118Z","iopub.status.idle":"2024-03-27T11:01:17.298400Z","shell.execute_reply.started":"2024-03-27T10:59:59.042082Z","shell.execute_reply":"2024-03-27T11:01:17.297497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint(USE_WAVELET_RAW_EEG)\nif USE_WAVELET_RAW_EEG:\n    print(1)\n    all_middle_raw_eegs = np.load(os.path.join(root_path, 'input','raw-eeg-dataset-middle', f'eeg_raw_middle_{USE_WAVELET_RAW_EEG}.npy'),allow_pickle=True).item()\nelse:\n    print(2)\n    all_middle_raw_eegs = np.load(os.path.join(root_path, 'input','raw-eeg-dataset-middle', 'eeg_raw_middle.npy'),allow_pickle=True).item()\n\nscale_dic = {\n\"0\": [np.min(all_middle_raw_eegs['middle_signal'][:,:,0]), np.max(all_middle_raw_eegs['middle_signal'][:,:,0])],\n\"1\": [np.min(all_middle_raw_eegs['middle_signal'][:,:,1]), np.max(all_middle_raw_eegs['middle_signal'][:,:,1])],\n\"2\": [np.min(all_middle_raw_eegs['middle_signal'][:,:,2]), np.max(all_middle_raw_eegs['middle_signal'][:,:,2])],\n\"3\": [np.min(all_middle_raw_eegs['middle_signal'][:,:,3]), np.max(all_middle_raw_eegs['middle_signal'][:,:,3])],\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:17.301785Z","iopub.execute_input":"2024-03-27T11:01:17.302073Z","iopub.status.idle":"2024-03-27T11:01:37.353282Z","shell.execute_reply.started":"2024-03-27T11:01:17.302048Z","shell.execute_reply":"2024-03-27T11:01:37.352326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(os.path.join(root_path, 'input','hms-harmful-brain-activity-classification', 'train.csv'))\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:37.354541Z","iopub.execute_input":"2024-03-27T11:01:37.354905Z","iopub.status.idle":"2024-03-27T11:01:37.672648Z","shell.execute_reply.started":"2024-03-27T11:01:37.354872Z","shell.execute_reply":"2024-03-27T11:01:37.671744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n\ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:37.674222Z","iopub.execute_input":"2024-03-27T11:01:37.674582Z","iopub.status.idle":"2024-03-27T11:01:37.810754Z","shell.execute_reply.started":"2024-03-27T11:01:37.674550Z","shell.execute_reply":"2024-03-27T11:01:37.809618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### make testset","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(os.path.join(root_path, 'input/hms-harmful-brain-activity-classification/test.csv'))\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:37.812231Z","iopub.execute_input":"2024-03-27T11:01:37.812613Z","iopub.status.idle":"2024-03-27T11:01:37.830135Z","shell.execute_reply.started":"2024-03-27T11:01:37.812583Z","shell.execute_reply":"2024-03-27T11:01:37.829188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ ALL SPECTROGRAMS\nPATH2 = os.path.join(root_path, 'input/hms-harmful-brain-activity-classification/test_spectrograms/')\nfiles2 = os.listdir(PATH2)\nprint(f'There are {len(files2)} test spectrogram parquets')\n\nspectrograms2 = {}\nfor i,f in enumerate(files2):\n    if i%100==0: print(i,', ',end='')\n    tmp = pd.read_parquet(f'{PATH2}{f}')\n    name = int(f.split('.')[0])\n    spectrograms2[name] = tmp.iloc[:,1:].values\n\n# RENAME FOR DATALOADER\ntest = test.rename({'spectrogram_id':'spec_id'},axis=1)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:37.831218Z","iopub.execute_input":"2024-03-27T11:01:37.831496Z","iopub.status.idle":"2024-03-27T11:01:38.154538Z","shell.execute_reply.started":"2024-03-27T11:01:37.831473Z","shell.execute_reply":"2024-03-27T11:01:38.153483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt, librosa\n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):\n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n\n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n\n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n\n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n\n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n\n        for kk in range(4):\n\n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256,\n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40\n            img[:,:,k] += mel_spec_db\n\n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n\n    #     if display:\n    #         plt.subplot(2,2,k+1)\n    #         plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n    #         plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n\n    # if display:\n    #     plt.show()\n    #     plt.figure(figsize=(10,5))\n    #     offset = 0\n    #     for k in range(4):\n    #         if k>0: offset -= signals[3-k].min()\n    #         plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n    #         offset += signals[3-k].max()\n    #     plt.legend()\n    #     plt.title(f'EEG {eeg_id} Signals')\n    #     plt.show()\n    #     print(); print('#'*25); print()\n\n    return img\n\ndef raw_eeg_middle(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    # img = np.zeros((128,256,4),dtype='float32')\n    \n    # if display: plt.figure(figsize=(10,7))\n    signals = np.zeros((10000,4),dtype='float32')\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET_RAW_EEG:\n                x = denoise(x, wavelet=USE_WAVELET_RAW_EEG)\n            # signals.append(x)\n            signals[:,k] = x\n\n    # if display: \n    #     plt.show()\n    #     plt.figure(figsize=(10,5))\n    #     offset = 0\n    #     for k in range(4):\n    #         if k>0: offset -= signals[3-k].min()\n    #         plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n    #         offset += signals[3-k].max()\n    #     plt.legend()\n    #     plt.title(f'EEG {eeg_id} Signals')\n    #     plt.show()\n    #     print(); print('#'*25); print()\n        \n    return signals","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:38.155841Z","iopub.execute_input":"2024-03-27T11:01:38.156124Z","iopub.status.idle":"2024-03-27T11:01:38.178972Z","shell.execute_reply.started":"2024-03-27T11:01:38.156099Z","shell.execute_reply":"2024-03-27T11:01:38.178102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ ALL EEG SPECTROGRAMS\nPATH2 = os.path.join(root_path, 'input/hms-harmful-brain-activity-classification/test_eegs/')\nDISPLAY = 1\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\nall_middel_eeg_arr_2 = np.zeros((len(EEG_IDS2),10000,4),dtype='float32')\n\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n\n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img\n\n    middle_signal = raw_eeg_middle(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_middel_eeg_arr_2[i,:,:] = middle_signal\n\nall_middle_eegs_2 = {}\nall_middle_eegs_2['eeg_id'] = EEG_IDS2\nall_middle_eegs_2['middle_signal'] = all_middel_eeg_arr_2","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:38.180105Z","iopub.execute_input":"2024-03-27T11:01:38.180460Z","iopub.status.idle":"2024-03-27T11:01:48.338690Z","shell.execute_reply.started":"2024-03-27T11:01:38.180428Z","shell.execute_reply":"2024-03-27T11:01:48.337413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train","metadata":{}},{"cell_type":"code","source":"## check data generator\ngen = DataGenerator(train,spectrograms,all_eegs,all_middle_raw_eegs,scale_dic, batch_size=16, shuffle=False)\nROWS=2; COLS=3; BATCHES=2\n\nfor i,(x,y) in enumerate(gen):\n    x_2d, x_1d = x[0], x[1]\n\n    plt.figure(figsize=[10,3])\n    plt.imshow(x_2d[0,:,:,:3])\n\n    plt.figure(figsize=[10,3])\n    for i in range(4):\n        plt.plot(x_1d[0,:,i])\n\n    break\ndel gen\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:48.346801Z","iopub.execute_input":"2024-03-27T11:01:48.352486Z","iopub.status.idle":"2024-03-27T11:01:49.351999Z","shell.execute_reply.started":"2024-03-27T11:01:48.352435Z","shell.execute_reply":"2024-03-27T11:01:49.351027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## check data generator\ngen = DataGenerator(test,spectrograms2,all_eegs2,all_middle_eegs_2,scale_dic, batch_size=16, shuffle=False, mode=\"test\")\nROWS=2; COLS=3; BATCHES=2\n\nfor i,(x,y) in enumerate(gen):\n    x_2d, x_1d = x[0], x[1]\n\n    plt.figure(figsize=[10,3])\n    plt.imshow(x_2d[0,:,:,:3])\n\n    plt.figure(figsize=[10,3])\n    for i in range(4):\n        plt.plot(x_1d[0,:,i])\n\n    break\ndel gen\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:49.353543Z","iopub.execute_input":"2024-03-27T11:01:49.353880Z","iopub.status.idle":"2024-03-27T11:01:50.191047Z","shell.execute_reply.started":"2024-03-27T11:01:49.353848Z","shell.execute_reply":"2024-03-27T11:01:50.190089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nLR_START = 1e-6\nLR_MAX = 1e-3\nLR_MIN = 1e-6\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS2 = 10\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS2 - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN\n    return lr\n\nrng = [i for i in range(EPOCHS2)]\nlr_y = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y, '-o')\nplt.xlabel('epoch',size=14); plt.ylabel('learning rate',size=14)\nplt.title('Cosine Training Schedule',size=16); plt.show()\n\nLR2 = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:50.192591Z","iopub.execute_input":"2024-03-27T11:01:50.193262Z","iopub.status.idle":"2024-03-27T11:01:50.478427Z","shell.execute_reply.started":"2024-03-27T11:01:50.193219Z","shell.execute_reply":"2024-03-27T11:01:50.477537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LR_START = 1e-4\nLR_MAX = 1e-3\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 1\nLR_STEP_DECAY = 0.1\nEVERY = 1\nEPOCHS = 4\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = LR_MAX * LR_STEP_DECAY**((epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS)//EVERY)\n    return lr\n\nrng = [i for i in range(EPOCHS)]\ny = [lrfn(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, y, 'o-');\nplt.xlabel('epoch',size=14); plt.ylabel('learning rate',size=14)\nplt.title('Step Training Schedule',size=16); plt.show()\n\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:50.479662Z","iopub.execute_input":"2024-03-27T11:01:50.479951Z","iopub.status.idle":"2024-03-27T11:01:50.765116Z","shell.execute_reply.started":"2024-03-27T11:01:50.479927Z","shell.execute_reply":"2024-03-27T11:01:50.764134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-index --find-links={root_path}/input/tf-efficientnet-whl-files {root_path}/input/tf-efficientnet-whl-files/efficientnet-1.1.1-py3-none-any.whl\n# !pip download efficientnet","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:01:50.766446Z","iopub.execute_input":"2024-03-27T11:01:50.766749Z","iopub.status.idle":"2024-03-27T11:02:04.429932Z","shell.execute_reply.started":"2024-03-27T11:01:50.766722Z","shell.execute_reply":"2024-03-27T11:02:04.428780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn\n\nclass MultimodalClassifier(tf.keras.Model): \n    def __init__(self): \n        super(MultimodalClassifier, self).__init__()\n\n        self.base_model = efn.EfficientNetB0(include_top=False, weights=None, input_shape=None)\n        self.base_model.load_weights(os.path.join(root_path, 'input','tf-efficientnet-imagenet-weights/efficientnet-b0_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5'))\n        \n\n\n        ### forward 2d input layer\n        self.forward_2d_layers = [\n            self.base_model,\n            tf.keras.layers.GlobalAveragePooling2D(),\n            # tf.keras.layers.Dense(512, activation='relu'),\n        ]\n        \n        ### forward 1d input layer\n        self.forward_1d_layers = [\n            tf.keras.layers.Conv1D(32, 3, activation='relu'),\n            tf.keras.layers.MaxPooling1D(3),\n            # tf.keras.layers.Conv1D(64, 3, activation='relu'),\n            # tf.keras.layers.MaxPooling1D(3),\n            tf.keras.layers.Flatten(),\n            tf.keras.layers.Dense(32, activation='relu', dtype='float32'),\n\n        ]\n        ### output layers\n        self.concat_layer_1 = tf.keras.layers.Concatenate()\n        self.output_layer = tf.keras.layers.Dense(6,activation='softmax', dtype='float32')\n    \n    def call(self, input: list): \n        input_2d, input_1d = input[0], input[1]\n\n\n        ### input preprocess ###\n        x1 = [input_2d[:,:,:,i:i+1] for i in range(4)]\n        x1 = tf.keras.layers.Concatenate(axis=1)(x1)\n        # EEG SPECTROGRAMS\n        x2 = [input_2d[:,:,:,i+4:i+5] for i in range(4)]\n        x2 = tf.keras.layers.Concatenate(axis=1)(x2)\n        # MAKE 512X512X3\n        if USE_KAGGLE_SPECTROGRAMS & USE_EEG_SPECTROGRAMS:\n            x = tf.keras.layers.Concatenate(axis=2)([x1,x2])\n        elif USE_EEG_SPECTROGRAMS: x = x2\n        else: x = x1\n        input_2d = tf.keras.layers.Concatenate(axis=3)([x,x,x])\n\n        \n        ## forward pass 2d\n        x_2 = input_2d\n        for layer in self.forward_2d_layers:\n            x_2 = layer(x_2)\n\n        ## forward pass 1d\n        x_1 = input_1d\n        for layer in self.forward_1d_layers:\n            x_1 = layer(x_1)\n\n        # output\n        output_1 = self.concat_layer_1([x_2, x_1])\n        output = self.output_layer(output_1)\n        # output = self.output_layer(x_2)\n        return output\n","metadata":{"execution":{"iopub.status.busy":"2024-03-27T11:02:04.431869Z","iopub.execute_input":"2024-03-27T11:02:04.432304Z","iopub.status.idle":"2024-03-27T11:02:04.462015Z","shell.execute_reply.started":"2024-03-27T11:02:04.432246Z","shell.execute_reply":"2024-03-27T11:02:04.461217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\nimport tensorflow.keras.backend as K, gc\n\nall_oof = []\nall_true = []\nn_fold = 5\ngkf = GroupKFold(n_splits=n_fold)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):  \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    \n    train_gen = DataGenerator(train.iloc[train_index],spectrograms,all_eegs,all_middle_raw_eegs,scale_dic, shuffle=True, batch_size=16, augment=False)\n    valid_gen = DataGenerator(train.iloc[valid_index],spectrograms,all_eegs,all_middle_raw_eegs,scale_dic, shuffle=False, batch_size=16, mode='valid')\n    \n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    K.clear_session()\n    with strategy.scope():\n        model = MultimodalClassifier()\n        opt = tf.keras.optimizers.legacy.Adam(learning_rate = 1e-3)\n        loss = tf.keras.losses.KLDivergence()\n        model.compile(loss=loss, optimizer = opt)\n    if LOAD_MODELS_FROM is None:\n        model.fit(train_gen, verbose=1,\n              validation_data = valid_gen,\n              epochs=EPOCHS, callbacks = [LR])\n        model.save_weights(f'EffNet_v{VER}_f{i}.h5')\n    else:\n        model.load_weights(f'{LOAD_MODELS_FROM}EffNet_v{VER}_f{i}.h5')\n        \n    oof = model.predict(valid_gen, verbose=1)\n    all_oof.append(oof)\n    all_true.append(train.iloc[valid_index][TARGETS].values)\n    \n    del model, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/kaggle-kl-div')\nfrom kaggle_kl_div import score\n\noof = pd.DataFrame(all_oof.copy())\noof['id'] = np.arange(len(oof))\n\ntrue = pd.DataFrame(all_true.copy())\ntrue['id'] = np.arange(len(true))\n\ncv = score(solution=true, submission=oof, row_id_column_name='id')\nprint('CV Score KL-Div for EfficientNetB2 =',cv)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### prediction","metadata":{}},{"cell_type":"code","source":"# INFER EFFICIENTNET ON TEST\npreds = []\nmodel = MultimodalClassifier()\ntest_gen = DataGenerator(test,spectrograms2,all_eegs2,all_middle_eegs_2,scale_dic, shuffle=False, batch_size=64, mode='test')\n\nfor i,(x,y) in enumerate(test_gen):\n    x_2d, x_1d = x[0], x[1]\n    model(x)\n    break\n\nfor i in range(n_fold):\n    print(f'Fold {i+1}')\n    if LOAD_MODELS_FROM:\n        model.load_weights(f'{LOAD_MODELS_FROM}EffNet_v{VER}_f{i}.h5')\n    else:\n        model.load_weights(f'EffNet_v{VER}_f{i}.h5')\n    pred = model.predict(test_gen, verbose=1)\n    preds.append(pred)\npred = np.mean(preds,axis=0)\nprint()\nprint('Test preds shape',pred.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}