{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7402356,"sourceType":"datasetVersion","datasetId":4304475},{"sourceId":7450712,"sourceType":"datasetVersion","datasetId":4336944},{"sourceId":158958765,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Library\n","metadata":{}},{"cell_type":"code","source":"import os, gc, glob\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport tensorflow as tf\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\n\nVER = 1","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:25.419234Z","iopub.execute_input":"2024-03-22T13:27:25.420220Z","iopub.status.idle":"2024-03-22T13:27:40.982865Z","shell.execute_reply.started":"2024-03-22T13:27:25.420157Z","shell.execute_reply":"2024-03-22T13:27:40.981872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load training data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:40.984840Z","iopub.execute_input":"2024-03-22T13:27:40.986325Z","iopub.status.idle":"2024-03-22T13:27:41.376653Z","shell.execute_reply.started":"2024-03-22T13:27:40.986284Z","shell.execute_reply":"2024-03-22T13:27:41.375477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate own spectrograms","metadata":{}},{"cell_type":"code","source":"READ_SPEC_FILES = False # If READ_SPEC_FILES is False, the code reads the combined file instead of individual files.","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.378263Z","iopub.execute_input":"2024-03-22T13:27:41.379037Z","iopub.status.idle":"2024-03-22T13:27:41.385418Z","shell.execute_reply.started":"2024-03-22T13:27:41.378992Z","shell.execute_reply":"2024-03-22T13:27:41.383902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.387053Z","iopub.execute_input":"2024-03-22T13:27:41.387408Z","iopub.status.idle":"2024-03-22T13:27:41.542568Z","shell.execute_reply.started":"2024-03-22T13:27:41.387374Z","shell.execute_reply":"2024-03-22T13:27:41.541374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['eeg_id'].head()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.546045Z","iopub.execute_input":"2024-03-22T13:27:41.546409Z","iopub.status.idle":"2024-03-22T13:27:41.556733Z","shell.execute_reply.started":"2024-03-22T13:27:41.546379Z","shell.execute_reply":"2024-03-22T13:27:41.555076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = df.groupby(['eeg_id','expert_consensus']).agg({'eeg_id': 'first','eeg_label_offset_seconds': 'min'})\ntrain.columns = ['eegs_id','eeg_offset']\n\n\n# add patient_id to train\ntmp = df.groupby(['eeg_id','expert_consensus'])[['patient_id','label_id', 'spectrogram_id']].agg({'patient_id' :'first',\n                                                                                      'spectrogram_id' : 'first', 'label_id':'first'})\ntrain[['patient_id','spectrogram_id', 'label_id']] = tmp\n\n# add expert votes\ntmp = df.groupby(['eeg_id','expert_consensus'])[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n\n# create label distribution\ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\n# add expert_consensus to train\ntmp = df.groupby(['eeg_id','expert_consensus'])[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.558197Z","iopub.execute_input":"2024-03-22T13:27:41.559092Z","iopub.status.idle":"2024-03-22T13:27:41.758442Z","shell.execute_reply.started":"2024-03-22T13:27:41.558909Z","shell.execute_reply":"2024-03-22T13:27:41.757206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eegs_id=train.loc[train['label_id'] == 1825637311]['eeg_id'].iloc[0]\nprint(eegs_id)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.760244Z","iopub.execute_input":"2024-03-22T13:27:41.760607Z","iopub.status.idle":"2024-03-22T13:27:41.769634Z","shell.execute_reply.started":"2024-03-22T13:27:41.760577Z","shell.execute_reply":"2024-03-22T13:27:41.768286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"directory_path = 'EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.770862Z","iopub.execute_input":"2024-03-22T13:27:41.771273Z","iopub.status.idle":"2024-03-22T13:27:41.777419Z","shell.execute_reply.started":"2024-03-22T13:27:41.771241Z","shell.execute_reply":"2024-03-22T13:27:41.776530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Denoising with Wavelet transform","metadata":{}},{"cell_type":"code","source":"import pywt\nprint(\"The wavelet functions we can use:\")\nprint(pywt.wavelist())\n\nUSE_WAVELET = 'db4'","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.778500Z","iopub.execute_input":"2024-03-22T13:27:41.778863Z","iopub.status.idle":"2024-03-22T13:27:41.813207Z","shell.execute_reply.started":"2024-03-22T13:27:41.778832Z","shell.execute_reply":"2024-03-22T13:27:41.811883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.814923Z","iopub.execute_input":"2024-03-22T13:27:41.815366Z","iopub.status.idle":"2024-03-22T13:27:41.825251Z","shell.execute_reply.started":"2024-03-22T13:27:41.815324Z","shell.execute_reply":"2024-03-22T13:27:41.823914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generating with librosa","metadata":{}},{"cell_type":"code","source":"import librosa\n\ndef spectrogram_from_eeg(parquet_path, display=False, offset=0):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    #middle = (len(eeg)-10_000)//2\n    #eeg = eeg.iloc[middle:middle+10_000]\n    off = int(200*offset)\n    eeg = eeg.iloc[off:off+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eegs_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eegs_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:28:33.516529Z","iopub.execute_input":"2024-03-22T13:28:33.517014Z","iopub.status.idle":"2024-03-22T13:28:33.535417Z","shell.execute_reply.started":"2024-03-22T13:28:33.516964Z","shell.execute_reply":"2024-03-22T13:28:33.534299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Apply to EEG","metadata":{}},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\ndirectory_path = 'EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:41.857874Z","iopub.execute_input":"2024-03-22T13:27:41.858407Z","iopub.status.idle":"2024-03-22T13:27:41.869185Z","shell.execute_reply.started":"2024-03-22T13:27:41.858361Z","shell.execute_reply":"2024-03-22T13:27:41.868043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\nEEG_IDS = train.label_id.unique()\nall_eegs = {}\n\n\nfor i,label_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='')\n    offset=train.loc[train['label_id'] == label_id]['eeg_offset'].iloc[0]\n    eegs_id=train.loc[train['label_id'] == label_id]['eeg_id'].iloc[0]\n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH}{eegs_id}.parquet', i<DISPLAY, offset)\n    \n    # SAVE TO DISK\n    if i==DISPLAY:\n        print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n    np.savez_compressed(f'{directory_path}{label_id}',img)    \n    img=None\n    gc.collect()\n   \n","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:28:35.162268Z","iopub.execute_input":"2024-03-22T13:28:35.163100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:59.386574Z","iopub.execute_input":"2024-03-22T13:27:59.387549Z","iopub.status.idle":"2024-03-22T13:27:59.885836Z","shell.execute_reply.started":"2024-03-22T13:27:59.387512Z","shell.execute_reply":"2024-03-22T13:27:59.884597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_eegs = {}\nfor i,e in enumerate(train.label_id.values):\n    if i%100==0: print(i,', ',end='')\n    x = np.load(f'{directory_path}{e}.npz', allow_pickle=True)\n    all_eegs[e] = x['arr_0']\n    x=None\n    gc.collect()\nnp.save('all_eegs', all_eegs)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T13:27:59.887464Z","iopub.execute_input":"2024-03-22T13:27:59.887833Z","iopub.status.idle":"2024-03-22T13:28:00.368171Z","shell.execute_reply.started":"2024-03-22T13:27:59.887802Z","shell.execute_reply":"2024-03-22T13:28:00.366614Z"},"trusted":true},"execution_count":null,"outputs":[]}]}