{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":160875016,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Base notebook: https://www.kaggle.com/code/cdeotte/catboost-starter-lb-0-60","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nimport time\n\nimport os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:37:16.396405Z","iopub.execute_input":"2024-01-29T14:37:16.397024Z","iopub.status.idle":"2024-01-29T14:37:16.775619Z","shell.execute_reply.started":"2024-01-29T14:37:16.396989Z","shell.execute_reply":"2024-01-29T14:37:16.774799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = train_df.columns[-6:]\nprint('Train shape:', train_df.shape )\nprint('Targets', list(TARGETS))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:37:18.357656Z","iopub.execute_input":"2024-01-29T14:37:18.358509Z","iopub.status.idle":"2024-01-29T14:37:18.698662Z","shell.execute_reply.started":"2024-01-29T14:37:18.358478Z","shell.execute_reply":"2024-01-29T14:37:18.697682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df['eeg_sub_id'].unique() # to 742","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:07:39.219598Z","iopub.execute_input":"2024-01-29T14:07:39.220543Z","iopub.status.idle":"2024-01-29T14:07:39.225685Z","shell.execute_reply.started":"2024-01-29T14:07:39.220502Z","shell.execute_reply":"2024-01-29T14:07:39.224253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Non-Overlapping Eeg Id Train Data\ntrain = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min_spec']\n\ntmp = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max_spec'] = tmp\n\ntmp = train_df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = train_df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n\ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = train_df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\n\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:37:32.230028Z","iopub.execute_input":"2024-01-29T14:37:32.230929Z","iopub.status.idle":"2024-01-29T14:37:32.340874Z","shell.execute_reply.started":"2024-01-29T14:37:32.230892Z","shell.execute_reply":"2024-01-29T14:37:32.340074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nEEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:37:37.253743Z","iopub.execute_input":"2024-01-29T14:37:37.254094Z","iopub.status.idle":"2024-01-29T14:37:37.259096Z","shell.execute_reply.started":"2024-01-29T14:37:37.254063Z","shell.execute_reply":"2024-01-29T14:37:37.258081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL EEGS\ntrain_eegs = np.load('',allow_pickle=True).item()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS from Chris Deotte dataset\nspectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:37:41.162625Z","iopub.execute_input":"2024-01-29T14:37:41.163422Z","iopub.status.idle":"2024-01-29T14:38:53.888830Z","shell.execute_reply.started":"2024-01-29T14:37:41.163379Z","shell.execute_reply":"2024-01-29T14:38:53.887896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL EEG SPECTROGRAMS from Chris Deotte dataset\nall_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:38:53.890501Z","iopub.execute_input":"2024-01-29T14:38:53.890793Z","iopub.status.idle":"2024-01-29T14:40:56.918121Z","shell.execute_reply.started":"2024-01-29T14:38:53.890767Z","shell.execute_reply":"2024-01-29T14:40:56.917128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_COLS = pd.read_parquet(f'{SPEC_PATH}1000086677.parquet').columns[1:]\nprint(f'{len(SPEC_COLS)} columns')\nSPEC_COLS","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:40:56.919397Z","iopub.execute_input":"2024-01-29T14:40:56.919679Z","iopub.status.idle":"2024-01-29T14:40:57.349793Z","shell.execute_reply.started":"2024-01-29T14:40:56.919653Z","shell.execute_reply":"2024-01-29T14:40:57.348805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_COLS = pd.read_parquet(f'{EEG_PATH}1000913311.parquet').columns\nEEG_COLS","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:40:57.352328Z","iopub.execute_input":"2024-01-29T14:40:57.352731Z","iopub.status.idle":"2024-01-29T14:40:57.398766Z","shell.execute_reply.started":"2024-01-29T14:40:57.352691Z","shell.execute_reply":"2024-01-29T14:40:57.397917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SPEC FEATURES 10 MINUTE WINDOW \nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_10m' for c in SPEC_COLS]\n\n# SPEC FEATURES 20 SECOND WINDOW \nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_20s' for c in SPEC_COLS]\n\n# EEG FEATURES 50 SECOND WINDOW\n# FEATURES += [f'{c}_mean_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_std_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_min_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_25%_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_50%_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_75%_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_max_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_skew_50s' for c in EEG_COLS]\n# FEATURES += [f'{c}_kurtosis_50s' for c in EEG_COLS]\n\n# EEG FEATURES 10 SECOND WINDOW\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ')","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:40:57.399749Z","iopub.execute_input":"2024-01-29T14:40:57.400023Z","iopub.status.idle":"2024-01-29T14:40:57.412200Z","shell.execute_reply.started":"2024-01-29T14:40:57.399998Z","shell.execute_reply":"2024-01-29T14:40:57.411310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.zeros((len(train),len(FEATURES)))  \ndata.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:40:57.413394Z","iopub.execute_input":"2024-01-29T14:40:57.413677Z","iopub.status.idle":"2024-01-29T14:40:57.423377Z","shell.execute_reply.started":"2024-01-29T14:40:57.413651Z","shell.execute_reply":"2024-01-29T14:40:57.422424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# row = train.iloc[0]\n# np.nanpercentile(spectrograms[row.spec_id], 25, axis=0)\n# from scipy.stats import skew, kurtosis\n# x = skew(train_eegs[row.spec_id][r+145:r+155,:],axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:19:40.814890Z","iopub.execute_input":"2024-01-29T14:19:40.815444Z","iopub.status.idle":"2024-01-29T14:19:40.821358Z","shell.execute_reply.started":"2024-01-29T14:19:40.815384Z","shell.execute_reply":"2024-01-29T14:19:40.820162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# ENGINEER FEATURES\nfor k in range(len(train)):\n    if k%100==0: print(k,', ',end='')\n    row = train.iloc[k]\n    r = int( (row['min_spec'] + row['max_spec'])//4 ) \n\n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,:400] = x\n    x = np.nanstd(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,400:800] = x\n    x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,800:1200] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 25, axis=0)\n    data[k,1200:1600] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 50, axis=0)\n    data[k,1600:2000] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 75, axis=0)\n    data[k,2000:2400] = x\n    x = np.nanmax(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,2400:2800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,2800:3200] = x\n    x = np.nanstd(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,3200:3600] = x\n    x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,3600:4000] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 25, axis=0)\n    data[k,4000:4400] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 50, axis=0)\n    data[k,4400:4800] = x\n    x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 75, axis=0)\n    data[k,4800:5200] = x\n    x = np.nanmax(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,5200:5600] = x\n\n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,5600:6112] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6112:6624] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6624:7136] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,7136:7648] = x\n\ntrain[FEATURES] = data\nprint(); print('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T14:42:35.769921Z","iopub.execute_input":"2024-01-29T14:42:35.770667Z","iopub.status.idle":"2024-01-29T16:15:57.915517Z","shell.execute_reply.started":"2024-01-29T14:42:35.770632Z","shell.execute_reply":"2024-01-29T16:15:57.914458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_eegs, spectrograms, data\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[:1]","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:19:40.388205Z","iopub.execute_input":"2024-01-29T16:19:40.388635Z","iopub.status.idle":"2024-01-29T16:19:40.475832Z","shell.execute_reply.started":"2024-01-29T16:19:40.388600Z","shell.execute_reply":"2024-01-29T16:19:40.474812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:19:54.476910Z","iopub.execute_input":"2024-01-29T16:19:54.477308Z","iopub.status.idle":"2024-01-29T16:19:55.145855Z","shell.execute_reply.started":"2024-01-29T16:19:54.477275Z","shell.execute_reply":"2024-01-29T16:19:55.144741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"import catboost as cat\nfrom catboost import CatBoostClassifier, Pool\n\nprint('CatBoost version',cat.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:20:48.147192Z","iopub.execute_input":"2024-01-29T16:20:48.148016Z","iopub.status.idle":"2024-01-29T16:20:49.268163Z","shell.execute_reply.started":"2024-01-29T16:20:48.147979Z","shell.execute_reply":"2024-01-29T16:20:49.267048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\n\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    \n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n\nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:25:47.033615Z","iopub.execute_input":"2024-01-29T16:25:47.034616Z","iopub.status.idle":"2024-01-29T16:43:28.145405Z","shell.execute_reply.started":"2024-01-29T16:25:47.034576Z","shell.execute_reply":"2024-01-29T16:43:28.144429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CV","metadata":{}},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/kaggle-kl-div')\nfrom kaggle_kl_div import score\n\noof = pd.DataFrame(all_oof.copy())\noof['id'] = np.arange(len(oof))\n\ntrue = pd.DataFrame(all_true.copy())\ntrue['id'] = np.arange(len(true))\n\ncv = score(solution=true, submission=oof, row_id_column_name='id')\nprint('CV Score KL-Div for CatBoost =',cv)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:43:45.211722Z","iopub.execute_input":"2024-01-29T16:43:45.212707Z","iopub.status.idle":"2024-01-29T16:43:45.308772Z","shell.execute_reply.started":"2024-01-29T16:43:45.212665Z","shell.execute_reply":"2024-01-29T16:43:45.307684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOP = 25\n\nfeature_importance = model.feature_importances_\nsorted_idx = np.argsort(feature_importance)\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], np.array(FEATURES)[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:44:15.065041Z","iopub.execute_input":"2024-01-29T16:44:15.066006Z","iopub.status.idle":"2024-01-29T16:44:15.707519Z","shell.execute_reply.started":"2024-01-29T16:44:15.065931Z","shell.execute_reply":"2024-01-29T16:44:15.706562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:46:02.964949Z","iopub.execute_input":"2024-01-29T16:46:02.965702Z","iopub.status.idle":"2024-01-29T16:46:02.988311Z","shell.execute_reply.started":"2024-01-29T16:46:02.965667Z","shell.execute_reply":"2024-01-29T16:46:02.986931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt, librosa\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:46:18.533693Z","iopub.execute_input":"2024-01-29T16:46:18.534760Z","iopub.status.idle":"2024-01-29T16:46:18.759090Z","shell.execute_reply.started":"2024-01-29T16:46:18.534722Z","shell.execute_reply":"2024-01-29T16:46:18.757957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 0\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:46:56.187534Z","iopub.execute_input":"2024-01-29T16:46:56.188281Z","iopub.status.idle":"2024-01-29T16:47:07.376405Z","shell.execute_reply.started":"2024-01-29T16:46:56.188244Z","shell.execute_reply":"2024-01-29T16:47:07.374867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n\n# ENGINEER FEATURES\nfor k in range(len(test)):\n    row = test.iloc[k]\n    r = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{r}.parquet')\n\n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean(spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanstd(spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n    x = np.nanmin(spec.iloc[:,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 25, axis=0)\n    data[k,1200:1600] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 50, axis=0)\n    data[k,1600:2000] = x\n    x = np.nanpercentile(spec.iloc[:,1:].values, 75, axis=0)\n    data[k,2000:2400] = x\n    x = np.nanmax(spec.iloc[:,1:].values, axis=0)\n    data[k,2400:2800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean(spec.iloc[145:155,1:].values, axis=0)\n    data[k,2800:3200] = x\n    x = np.nanstd(spec.iloc[145:155,1:].values, axis=0)\n    data[k,3200:3600] = x\n    x = np.nanmin(spec.iloc[145:155,1:].values, axis=0)\n    data[k,3600:4000] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 25, axis=0)\n    data[k,4000:4400] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 50, axis=0)\n    data[k,4400:4800] = x\n    x = np.nanpercentile(spec.iloc[145:155,1:].values, 75, axis=0)\n    data[k,4800:5200] = x\n    x = np.nanmax(spec.iloc[145:155,1:].values, axis=0)\n    data[k,5200:5600] = x\n\n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs2[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,5600:6112] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6112:6624] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,6624:7136] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,7136:7648] = x\n\ntest[FEATURES] = data\nprint(); print('New test shape:',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:55:03.373377Z","iopub.execute_input":"2024-01-29T16:55:03.374158Z","iopub.status.idle":"2024-01-29T16:55:17.351645Z","shell.execute_reply.started":"2024-01-29T16:55:03.374119Z","shell.execute_reply":"2024-01-29T16:55:17.350592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:56:58.286670Z","iopub.execute_input":"2024-01-29T16:56:58.287628Z","iopub.status.idle":"2024-01-29T16:56:58.339384Z","shell.execute_reply.started":"2024-01-29T16:56:58.287589Z","shell.execute_reply":"2024-01-29T16:56:58.338302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sub","metadata":{}},{"cell_type":"code","source":"# INFER CATBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i,', ',end='')\n    model = CatBoostClassifier(task_type='GPU')\n    model.load_model(f'CAT_f{i}.cat')\n    \n    test_pool = Pool(\n        data = test[FEATURES]\n    )\n    \n    pred = model.predict_proba(test_pool)\n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:58:41.701143Z","iopub.execute_input":"2024-01-29T16:58:41.701544Z","iopub.status.idle":"2024-01-29T16:58:46.775177Z","shell.execute_reply.started":"2024-01-29T16:58:41.701512Z","shell.execute_reply":"2024-01-29T16:58:46.774334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.mean(preds,axis=0)\nprint(pred)\nprint('Test preds shape',pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:59:01.665159Z","iopub.execute_input":"2024-01-29T16:59:01.665984Z","iopub.status.idle":"2024-01-29T16:59:01.672363Z","shell.execute_reply.started":"2024-01-29T16:59:01.665941Z","shell.execute_reply":"2024-01-29T16:59:01.671474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:59:10.958794Z","iopub.execute_input":"2024-01-29T16:59:10.959556Z","iopub.status.idle":"2024-01-29T16:59:10.976143Z","shell.execute_reply.started":"2024-01-29T16:59:10.959517Z","shell.execute_reply":"2024-01-29T16:59:10.975172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:59:32.259702Z","iopub.execute_input":"2024-01-29T16:59:32.260092Z","iopub.status.idle":"2024-01-29T16:59:32.286787Z","shell.execute_reply.started":"2024-01-29T16:59:32.260062Z","shell.execute_reply":"2024-01-29T16:59:32.285602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T16:59:46.468811Z","iopub.execute_input":"2024-01-29T16:59:46.469542Z","iopub.status.idle":"2024-01-29T16:59:46.478407Z","shell.execute_reply.started":"2024-01-29T16:59:46.469504Z","shell.execute_reply":"2024-01-29T16:59:46.477400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}