{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8006040,"sourceType":"datasetVersion","datasetId":4603929}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"An execution error will occur due to insufficient output size, so please change to `FILE_OUTPUT = True` to output to a files.","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\n\nOUTPUT_SPECTROGRAM = False\nOUTPUT_EEG = False\nOUTPUT_EEG_STANDARDIZE = False\nOUTPUT_EEG_DENOISE = False\nOUTPUT_RAW_EEG = False\nOUTPUT_SAMPLES = True\n\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T08:04:02.244437Z","iopub.execute_input":"2024-04-03T08:04:02.245258Z","iopub.status.idle":"2024-04-03T08:04:04.047177Z","shell.execute_reply.started":"2024-04-03T08:04:02.245216Z","shell.execute_reply":"2024-04-03T08:04:04.045978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T08:04:04.049372Z","iopub.execute_input":"2024-04-03T08:04:04.050524Z","iopub.status.idle":"2024-04-03T08:04:04.058326Z","shell.execute_reply.started":"2024-04-03T08:04:04.050482Z","shell.execute_reply":"2024-04-03T08:04:04.056964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Spectrograms\n","metadata":{}},{"cell_type":"code","source":"%%time\n\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} train spectrogram parquets')\nif OUTPUT_SPECTROGRAM: \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\n\n    # SAVE EEG SPECTROGRAM DICTIONARY\n    np.save('train-spectrograms',spectrograms)\n\n    del spectrograms","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.331061Z","iopub.execute_input":"2024-04-02T13:05:21.331506Z","iopub.status.idle":"2024-04-02T13:05:21.588035Z","shell.execute_reply.started":"2024-04-02T13:05:21.331456Z","shell.execute_reply":"2024-04-02T13:05:21.586801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train EEGs ","metadata":{}},{"cell_type":"code","source":"# !pip install pywt\n# import pywt\n# print(\"The wavelet functions we can use:\")\n# # print(pywt.wavelist())","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.590234Z","iopub.execute_input":"2024-04-02T13:05:21.590551Z","iopub.status.idle":"2024-04-02T13:05:21.596483Z","shell.execute_reply.started":"2024-04-02T13:05:21.590524Z","shell.execute_reply":"2024-04-02T13:05:21.595124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    maddest = np.mean(np.absolute(coeff[-level] - np.mean(coeff[-level])))\n    sigma = (1/0.6745) * maddest\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.597423Z","iopub.execute_input":"2024-04-02T13:05:21.597732Z","iopub.status.idle":"2024-04-02T13:05:21.611760Z","shell.execute_reply.started":"2024-04-02T13:05:21.597706Z","shell.execute_reply":"2024-04-02T13:05:21.610617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\ndef spectrogram_from_eeg(parquet_path, USE_WAVELET=None, STANDARDIZE = False,\n                         col1 = 100, col2 = 300, col3 = 30):\n        \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((col1,col2,4),dtype='float32')\n    \n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n            x1 = eeg[COLS[kk]].values\n            x2 = eeg[COLS[kk+1]].values\n            m = np.nanmean(x1)\n            if np.isnan(x1).mean()<1: x1 = np.nan_to_num(x1,nan=m)\n            else: x1[:] = 0\n            m = np.nanmean(x2)\n            if np.isnan(x2).mean()<1: x2 = np.nan_to_num(x2,nan=m)\n            else: x2[:] = 0\n                \n            # COMPUTE PAIR DIFFERENCES\n            x = x1 - x2\n            \n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x) // col2, \n                  n_fft=1024, n_mels=col1, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1] // col3) * col3\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            if STANDARDIZE: mel_spec_db = (mel_spec_db+40) / 40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.613004Z","iopub.execute_input":"2024-04-02T13:05:21.613402Z","iopub.status.idle":"2024-04-02T13:05:21.635343Z","shell.execute_reply.started":"2024-04-02T13:05:21.613371Z","shell.execute_reply":"2024-04-02T13:05:21.634395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\nEEG_IDS = train.eeg_id.unique()\neegs = {}\n\nif OUTPUT_EEG: \n    for i,eeg_id in enumerate(EEG_IDS):\n        if (i%100==0)&(i!=0): print(i,', ',end='')\n\n        # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n        eeg = pd.read_parquet(f'{PATH}{eeg_id}.parquet')\n        middle = (len(eeg)-10_000)//2\n        eeg = eeg.iloc[middle:middle+10_000]\n\n        # CREATE SPECTROGRAM FROM EEG PARQUET\n        eegs[eeg_id] = spectrogram_from_eeg(eeg)\n        \n        # SAVE TO DISK\n        if i==4: print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n\n    # SAVE EEG SPECTROGRAM DICTIONARY\n    np.save('train-eegs',eegs)\n\n    del eegs","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.636448Z","iopub.execute_input":"2024-04-02T13:05:21.637513Z","iopub.status.idle":"2024-04-02T13:05:21.661638Z","shell.execute_reply.started":"2024-04-02T13:05:21.637480Z","shell.execute_reply":"2024-04-02T13:05:21.660617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\neegs_std = {}\nif OUTPUT_EEG_STANDARDIZE: \n    for i,eeg_id in enumerate(EEG_IDS):\n        if (i%100==0)&(i!=0): print(i,', ',end='')\n\n        # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n        eeg = pd.read_parquet(f'{PATH}{eeg_id}.parquet')\n        middle = (len(eeg)-10_000)//2\n        eeg = eeg.iloc[middle:middle+10_000]\n\n        # CREATE SPECTROGRAM FROM EEG PARQUET\n        eegs_std[eeg_id] = spectrogram_from_eeg(eeg, STANDARDIZE = True, col1 = 128, col2 = 256, col3 = 32)\n        \n        # SAVE TO DISK\n        if i==4: print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n\n    # SAVE EEG SPECTROGRAM DICTIONARY\n    np.save('train-eegs_std',eegs_std)\n    \n    del eegs_std","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.663001Z","iopub.execute_input":"2024-04-02T13:05:21.663935Z","iopub.status.idle":"2024-04-02T13:05:21.681760Z","shell.execute_reply.started":"2024-04-02T13:05:21.663897Z","shell.execute_reply":"2024-04-02T13:05:21.680533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train EEGs (Denoised)","metadata":{}},{"cell_type":"code","source":"%%time\n\nif OUTPUT_EEG_DENOISE :\n    eegs_denoise = {}\n    for i,eeg_id in enumerate(EEG_IDS):\n        if (i%100==0)&(i!=0): print(i,', ',end='')\n            \n        # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n        eeg = pd.read_parquet(f'{PATH}{eeg_id}.parquet')\n        middle = (len(eeg)-10_000)//2\n        eeg = eeg.iloc[middle:middle+10_000]\n\n        # CREATE SPECTROGRAM FROM EEG PARQUET\n        eegs_denoise[eeg_id] = spectrogram_from_eeg(eeg, USE_WAVELET=\"db8\", STANDARDIZE = True,\n                                                    col1 = 128, col2 = 256, col3 = 32)\n\n        # SAVE TO DISK\n        if i==4: print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n\n    # SAVE EEG SPECTROGRAM DICTIONARY\n    np.save('train-eegs_denoise', eegs_denoise)\n    \n    del eegs_denoise","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.683412Z","iopub.execute_input":"2024-04-02T13:05:21.684601Z","iopub.status.idle":"2024-04-02T13:05:21.699177Z","shell.execute_reply.started":"2024-04-02T13:05:21.684544Z","shell.execute_reply":"2024-04-02T13:05:21.697806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Raw EEGs","metadata":{}},{"cell_type":"code","source":"FEATS2 = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\nFEAT2IDX = {x:y for x,y in zip(FEATS2,range(len(FEATS2)))}\n\ndef eeg_from_parquet(eeg):\n    rows = len(eeg)\n    offset = (rows-10_000) // 2\n    eeg = eeg.iloc[offset : offset+10_000]\n    data = np.zeros((10_000, len(FEATS2)))\n    \n    for j,col in enumerate(FEATS2):\n        x = eeg[col].values.astype('float32')\n        m = np.nanmean(x)\n        if np.isnan(x).mean() < 1 : x = np.nan_to_num(x,nan=m)\n        else: x[:] = 0\n        \n        data[:,j] = x\n\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.703136Z","iopub.execute_input":"2024-04-02T13:05:21.704254Z","iopub.status.idle":"2024-04-02T13:05:21.714962Z","shell.execute_reply.started":"2024-04-02T13:05:21.704210Z","shell.execute_reply":"2024-04-02T13:05:21.714139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\nEEG_IDS = train.eeg_id.unique()\n\nraw_eegs = {}\nif OUTPUT_RAW_EEG: \n    for i,eeg_id in enumerate(EEG_IDS):\n        if (i%100==0)&(i!=0): print(i,', ',end='')\n\n        eeg = pd.read_parquet(f'{PATH}{eeg_id}.parquet')\n        raw_eegs[eeg_id] = eeg_from_parquet(eeg)\n        \n        # SAVE TO DISK\n        if i==4: print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n\n    # SAVE EEG SPECTROGRAM DICTIONARY\n    np.save('train-raw_eegs',raw_eegs)\n    \n    del raw_eegs","metadata":{"execution":{"iopub.status.busy":"2024-04-02T13:05:21.716619Z","iopub.execute_input":"2024-04-02T13:05:21.717337Z","iopub.status.idle":"2024-04-02T13:05:21.731194Z","shell.execute_reply.started":"2024-04-02T13:05:21.717297Z","shell.execute_reply":"2024-04-02T13:05:21.729886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample","metadata":{}},{"cell_type":"code","source":"if OUTPUT_SAMPLES :\n    train_tmp = train.copy()\n    train_tmp = train_tmp.groupby('eeg_id')[['spectrogram_id', 'patient_id']].agg('first')\n    \n    tmp = train.groupby('eeg_id')[TARGETS].agg('sum')\n    y_data = tmp[TARGETS].values / tmp[TARGETS].values.sum(axis=1,keepdims=True)\n    train_tmp[TARGETS] = y_data\n    \n    train_tmp = train_tmp.reset_index()\n    \n    train_sample = train_tmp.sample(7500, random_state=123).reset_index(drop=True)\n    train_sample.to_csv('sample-train.csv',index=False)\n    \n    # Spectrograms\n    print(\"Create Spectrogram Sample Datas\")\n    sample_spectrogram_ids = train_sample.spectrogram_id.unique()\n    spectrograms = np.load('/kaggle/input/hms-brain-wave-activity-eegs-spectrograms/train-spectrograms.npy',allow_pickle=True).item()\n    \n    sample_spectrograms = {}\n    for spectrogram_id in sample_spectrogram_ids :\n        sample_spectrograms[spectrogram_id] = spectrograms[spectrogram_id]\n    np.save('sample-train-spectrograms',sample_spectrograms)\n    del spectrograms\n    \n    # EEGs\n    print(\"Create EEG Sample Datas\")\n    sample_eeg_ids = train_sample.eeg_id.unique()\n    eegs = np.load('/kaggle/input/hms-brain-wave-activity-eegs-spectrograms/train-eegs.npy',allow_pickle=True).item()\n    \n    sample_eegs = {}\n    for eeg_id in sample_eeg_ids :\n        sample_eegs[eeg_id] = eegs[eeg_id]\n    np.save('sample-train-eegs',sample_eegs)\n    del eegs\n    \n    # Raw_EEGs\n    print(\"Create Raw EEG Sample Datas\")\n    raw_eegs = np.load('/kaggle/input/hms-brain-wave-activity-eegs-spectrograms/train-eegs_raw.npy',allow_pickle=True).item()\n    \n    sample_raw_eegs = {}\n    for eeg_id in sample_eeg_ids :\n        sample_raw_eegs[eeg_id] = raw_eegs[eeg_id]\n    np.save('sample-train-eegs_raw',sample_raw_eegs)\n    del raw_eegs","metadata":{"execution":{"iopub.status.busy":"2024-04-03T08:04:17.218044Z","iopub.execute_input":"2024-04-03T08:04:17.218453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Finish!\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}