{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":163545774,"sourceType":"kernelVersion"},{"sourceId":165852546,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Here we train three different gradient boosting models XGBoost, CatBoost and LightGBM as an ensemble.\n\n### Credits to York Yong for sharing his notebook so we can use it as our baseline (https://www.kaggle.com/code/yorkyong/exploring-eeg-a-beginner-s-guide)\n\n### Also thanks to Chris Deotte for his work on transforming eeg data to spectrograms.","metadata":{}},{"cell_type":"code","source":"# imports needed\nimport os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nimport optuna\nfrom sklearn.metrics import log_loss\nimport warnings\nwarnings.filterwarnings('ignore')\nimport catboost as cat\nfrom catboost import CatBoostClassifier, Pool\nprint('CatBoost version',cat.__version__)\nfrom sklearn.impute import SimpleImputer\nfrom sklearn import preprocessing\nfrom imblearn.over_sampling import SMOTE\nfrom skopt import BayesSearchCV\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold, GroupKFold\nimport xgboost as xgb","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"/kaggle/input/dataset\"))","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:28:56.580559Z","iopub.execute_input":"2024-03-03T12:28:56.580781Z","iopub.status.idle":"2024-03-03T12:28:56.593549Z","shell.execute_reply.started":"2024-03-03T12:28:56.580762Z","shell.execute_reply":"2024-03-03T12:28:56.592575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/dataset/dataset.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:28:56.880479Z","iopub.execute_input":"2024-03-03T12:28:56.881161Z","iopub.status.idle":"2024-03-03T12:30:22.480063Z","shell.execute_reply.started":"2024-03-03T12:28:56.881131Z","shell.execute_reply":"2024-03-03T12:30:22.478976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.482418Z","iopub.execute_input":"2024-03-03T12:30:22.483185Z","iopub.status.idle":"2024-03-03T12:30:22.489777Z","shell.execute_reply.started":"2024-03-03T12:30:22.483147Z","shell.execute_reply":"2024-03-03T12:30:22.488802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.491000Z","iopub.execute_input":"2024-03-03T12:30:22.491272Z","iopub.status.idle":"2024-03-03T12:30:22.517320Z","shell.execute_reply.started":"2024-03-03T12:30:22.491250Z","shell.execute_reply":"2024-03-03T12:30:22.516235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n# TARGETS = df.columns[-6:]\n# print('Train shape:', df.shape )\n# print('Targets', list(TARGETS))\n# df.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.519735Z","iopub.execute_input":"2024-03-03T12:30:22.520093Z","iopub.status.idle":"2024-03-03T12:30:22.524220Z","shell.execute_reply.started":"2024-03-03T12:30:22.520069Z","shell.execute_reply":"2024-03-03T12:30:22.523311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# print(df.shape)\n# rows_to_delete = []\n# for index, row in df.iterrows():\n    \n#     if (row.iloc[-6:-1].sum() < 2) or ((row.iloc[-6:-1].sum() > 1) and ((row.iloc[-6:-1].max() - row.iloc[-6:-1].min()) < 2)):\n        \n#         rows_to_delete.append(index)\n        \n        \n# df.drop(rows_to_delete, inplace=True)\n\n# print(df.shape)\n        \n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.525415Z","iopub.execute_input":"2024-03-03T12:30:22.525691Z","iopub.status.idle":"2024-03-03T12:30:22.535560Z","shell.execute_reply.started":"2024-03-03T12:30:22.525669Z","shell.execute_reply":"2024-03-03T12:30:22.534742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# train2 = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n#     {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\n# train2.columns = ['spec_id','min']\n\n# tmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n#     {'spectrogram_label_offset_seconds':'max'})\n# train['max'] = tmp\n\n# tmp = df.groupby('eeg_id')[['patient_id']].agg('first')\n# train2['patient_id'] = tmp\n\n# tmp = df.groupby('eeg_id')[TARGETS].agg('sum')\n# for t in TARGETS:\n#     train2[t] = tmp[t].values\n    \n# y_data = train2[TARGETS].values\n# y_data = y_data / y_data.sum(axis=1,keepdims=True)\n# train2[TARGETS] = y_data\n\n# print(df.groupby('eeg_id')[['expert_consensus']])\n\n# # COULD CHANGE THIS NOT TO BE FIRST, BUT GET THE HIGHEST NORMALIZED SCORE OF THE VOTE COLUMNS!!!\n# tmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\n# train2['target'] = tmp\n\n# train = train.reset_index()\n# print('Train non-overlapp eeg_id shape:', train2.shape )\n# train2.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.537204Z","iopub.execute_input":"2024-03-03T12:30:22.537805Z","iopub.status.idle":"2024-03-03T12:30:22.545699Z","shell.execute_reply.started":"2024-03-03T12:30:22.537781Z","shell.execute_reply":"2024-03-03T12:30:22.544946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n#     {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\n# train.columns = ['spec_id','min']\n\n# tmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n#     {'spectrogram_label_offset_seconds':'max'})\n# train['max'] = tmp\n\n# tmp = df.groupby('eeg_id')[['patient_id']].agg('first')\n# train['patient_id'] = tmp\n\n# tmp = df.groupby('eeg_id')[TARGETS].agg('sum')\n# for t in TARGETS:\n#     train[t] = tmp[t].values\n    \n# y_data = train[TARGETS].values\n# y_data = y_data / y_data.sum(axis=1,keepdims=True)\n# train[TARGETS] = y_data\n\n# # COULD CHANGE THIS NOT TO BE FIRST, BUT GET THE HIGHEST NORMALIZED SCORE OF THE VOTE COLUMNS!!!\n# tmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\n# train['target'] = tmp\n\n# train = train.reset_index()\n# print('Train non-overlapp eeg_id shape:', train.shape )\n# train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.546820Z","iopub.execute_input":"2024-03-03T12:30:22.547147Z","iopub.status.idle":"2024-03-03T12:30:22.560657Z","shell.execute_reply.started":"2024-03-03T12:30:22.547118Z","shell.execute_reply":"2024-03-03T12:30:22.559954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merged_df = pd.merge(train, train2, on='eeg_id', suffixes=('_df1', '_df2'))\n\n# different_rows = merged_df[merged_df['target_df1'] != merged_df['target_df2']]\n\n# result_df = different_rows[['eeg_id', 'seizure_vote_df1', 'seizure_vote_df2', 'lpd_vote_df1', 'lpd_vote_df2', 'gpd_vote_df1', 'gpd_vote_df2', 'lrda_vote_df1', 'lrda_vote_df2', 'grda_vote_df1', 'grda_vote_df2', 'other_vote_df1', 'other_vote_df2']]\n\n# result_df.columns = ['eeg_id', 'seizure_vote_df1', 'seizure_vote_df2', 'lpd_vote_df1', 'lpd_vote_df2', 'gpd_vote_df1', 'gpd_vote_df2', 'lrda_vote_df1', 'lrda_vote_df2', 'grda_vote_df1', 'grda_vote_df2', 'other_vote_df1', 'other_vote_df2']\n\n# print(result_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.561709Z","iopub.execute_input":"2024-03-03T12:30:22.561985Z","iopub.status.idle":"2024-03-03T12:30:22.569626Z","shell.execute_reply.started":"2024-03-03T12:30:22.561963Z","shell.execute_reply":"2024-03-03T12:30:22.568891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# %%time\n# # READ ALL SPECTROGRAMS (SWITCHED TO CHRIS DEOTTE's DATASET FOR THIS due to RAM problems)\n\n# # PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n# # files = os.listdir(PATH)\n# # print(f'There are {len(files)} spectrogram parquets')\n   \n# # spectrograms = {}\n# # for i,f in enumerate(files):\n# #     if i%100 == 0: \n        \n# #         print(i)\n        \n# #     tmp = pd.read_parquet(f'{PATH}{f}')\n# #     name = int(f.split('.')[0])\n# #     spectrograms[name] = tmp.iloc[:,1:].values\n\n# spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.570604Z","iopub.execute_input":"2024-03-03T12:30:22.570874Z","iopub.status.idle":"2024-03-03T12:30:22.583402Z","shell.execute_reply.started":"2024-03-03T12:30:22.570824Z","shell.execute_reply":"2024-03-03T12:30:22.582585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# %%time\n# PATHeeg = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n# files = os.listdir(PATHeeg)\n# print(f'There are {len(files)} eeg parquets')\n   \n# eeg = {}\n# for i,f in enumerate(files):\n#     if i%100==0: print(i,', ',end='')\n#     tmp = pd.read_parquet(f'{PATHeeg}{f}')\n#     name = int(f.split('.')[0])\n#     eeg[name] = tmp.iloc[:,1:].values","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.587068Z","iopub.execute_input":"2024-03-03T12:30:22.587768Z","iopub.status.idle":"2024-03-03T12:30:22.593638Z","shell.execute_reply.started":"2024-03-03T12:30:22.587726Z","shell.execute_reply":"2024-03-03T12:30:22.592772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(list(spectrograms.keys())[100])","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.594552Z","iopub.execute_input":"2024-03-03T12:30:22.594829Z","iopub.status.idle":"2024-03-03T12:30:22.603439Z","shell.execute_reply.started":"2024-03-03T12:30:22.594807Z","shell.execute_reply":"2024-03-03T12:30:22.602598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# %time\n\n# # ENGINEER FEATURES\n\n# PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n\n# # FEATURE NAMES\n# SPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\n\n# print(SPEC_COLS)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.604349Z","iopub.execute_input":"2024-03-03T12:30:22.604603Z","iopub.status.idle":"2024-03-03T12:30:22.613064Z","shell.execute_reply.started":"2024-03-03T12:30:22.604583Z","shell.execute_reply":"2024-03-03T12:30:22.612309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(len(train))","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.614140Z","iopub.execute_input":"2024-03-03T12:30:22.614416Z","iopub.status.idle":"2024-03-03T12:30:22.625635Z","shell.execute_reply.started":"2024-03-03T12:30:22.614394Z","shell.execute_reply":"2024-03-03T12:30:22.624793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# # READ ALL EEG\n\n# files = os.listdir(EEG_PATH)\n# print(f'There are {len(files)} eeg parquets')\n   \n# eeg = {}\n# for i,f in enumerate(files):\n#     if i%100==0:\n        \n#         print(i)\n        \n#     tmp = pd.read_parquet(f'{EEG_PATH}{f}')\n#     name = int(f.split('.')[0])\n#     eeg[name] = tmp.iloc[:,1:].values\n\n# # from chris deotte\n# all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()\n\n# EEG_COLS = pd.read_parquet(f'{EEG_PATH}1000913311.parquet').columns\n# EEG_COLS","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.626626Z","iopub.execute_input":"2024-03-03T12:30:22.626928Z","iopub.status.idle":"2024-03-03T12:30:22.635123Z","shell.execute_reply.started":"2024-03-03T12:30:22.626899Z","shell.execute_reply":"2024-03-03T12:30:22.634365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()\n\n# EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# EEG_COLS = pd.read_parquet(f'{EEG_PATH}1000913311.parquet').columns\n# EEG_COLS","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.636332Z","iopub.execute_input":"2024-03-03T12:30:22.636667Z","iopub.status.idle":"2024-03-03T12:30:22.643665Z","shell.execute_reply.started":"2024-03-03T12:30:22.636638Z","shell.execute_reply":"2024-03-03T12:30:22.642866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n# FEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_max_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_std_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_mdn_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_25%_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_50%_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_75%_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_max_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_std_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_mdn_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_25%_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_50%_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_75%_20s' for c in SPEC_COLS]\n\n# FEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\n# FEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\n# FEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\n# FEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\n\n# print(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\n# data = np.zeros((len(train),len(FEATURES)))\n# for k in range(len(train)):\n#     # print(k)\n#     if k%100==0: print(k)\n#     row = train.iloc[k]\n#     r = int( (row['min'] + row['max'])//4 ) \n    \n# #     print(row['min'])\n# #     print(row['max'])\n# #     print(r)\n\n#     # 10 MINUTE WINDOW FEATURES\n#     x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n#     data[k,:400] = x\n#     x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n#     data[k,400:800] = x\n#     x = np.nanmax(spectrograms[row.spec_id][r:r+300,:],axis=0)\n#     data[k,800:1200] = x\n#     x = np.nanstd(spectrograms[row.spec_id][r:r+300,:],axis=0)\n#     data[k,1200:1600] = x\n#     x = np.nanmedian(spectrograms[row.spec_id][r:r+300,:],axis=0)\n#     data[k,1600:2000] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 25, axis=0)\n#     data[k,2000:2400] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 50, axis=0)\n#     data[k,2400:2800] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r:r+300,:], 75, axis=0)\n#     data[k,2800:3200] = x\n\n\n#     # 20 SECOND WINDOW FEATURES\n#     x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n#     data[k,3200:3600] = x\n#     x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n#     data[k,3600:4000] = x\n#     x = np.nanmax(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n#     data[k,4000:4400] = x\n#     x = np.nanstd(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n#     data[k,4400:4800] = x\n#     x = np.nanmedian(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n#     data[k,4800:5200] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 25, axis=0)\n#     data[k,5200:5600] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 50, axis=0)\n#     data[k,5600:6000] = x\n#     x = np.nanpercentile(spectrograms[row.spec_id][r+145:r+155,:], 75, axis=0)\n#     data[k,6000:6400] = x\n    \n#     eeg_spec = np.zeros((512,256),dtype='float32')\n#     xx = all_eegs[row.eeg_id]\n#     for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n#     # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n#     x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,6400:6912] = x\n#     x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,6912:7424] = x\n#     x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,7424:7936] = x\n#     x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,7936:8448] = x\n    \n\n# #     # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n# #     eeg_spec = np.zeros((512,256),dtype='float32')\n# #     xx = eeg[row.eeg_id]\n# #     print(xx.shape)\n# #     for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n# #     # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n# #     x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n# #     data[k,1600:2112] = x\n# #     x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n# #     data[k,2112:2624] = x\n# #     x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n# #     data[k,2624:3136] = x\n# #     x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n# #     data[k,3136:3648] = x\n\n# train[FEATURES] = data\n# print(); print('New train shape:',train.shape)\n\n# # del spectrograms\n# # gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.645418Z","iopub.execute_input":"2024-03-03T12:30:22.645644Z","iopub.status.idle":"2024-03-03T12:30:22.655697Z","shell.execute_reply.started":"2024-03-03T12:30:22.645624Z","shell.execute_reply":"2024-03-03T12:30:22.654877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n#  EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# # READ ALL EEG\n\n# files = os.listdir(EEG_PATH)\n# print(f'There are {len(files)} eeg parquets')\n   \n# eeg = {}\n# for i,f in enumerate(files):\n#     if i%100==0:\n        \n#         print(i)\n        \n#     tmp = pd.read_parquet(f'{EEG_PATH}{f}')\n#     name = int(f.split('.')[0])\n#     eeg[name] = tmp.iloc[:,1:].values\n\n# from chris deotte\n# all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()\n\n# EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# EEG_COLS = pd.read_parquet(f'{EEG_PATH}1000913311.parquet').columns\n# EEG_COLS\n\n# for k in range(len(train)):\n#     # print(k)\n#     if k%100==0: print(k)\n    \n#     eeg_spec = np.zeros((512,256),dtype='float32')\n#     xx = all_eegs[row.eeg_id]\n#     for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n#     # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n#     x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,6400:6912] = x\n#     x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,6912:7424] = x\n#     x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,7424:7936] = x\n#     x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n#     data[k,7936:8448] = x\n    \n# train[FEATURES] = data\n# print(); print('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.656781Z","iopub.execute_input":"2024-03-03T12:30:22.657112Z","iopub.status.idle":"2024-03-03T12:30:22.668378Z","shell.execute_reply.started":"2024-03-03T12:30:22.657083Z","shell.execute_reply":"2024-03-03T12:30:22.667481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # FREE MEMORY\n# del all_eegs, spectrograms, data\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.669337Z","iopub.execute_input":"2024-03-03T12:30:22.669575Z","iopub.status.idle":"2024-03-03T12:30:22.680484Z","shell.execute_reply.started":"2024-03-03T12:30:22.669554Z","shell.execute_reply":"2024-03-03T12:30:22.679764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n# train.fillna(train.mean(), inplace=True)\n# nan_per_column = train.isnull().sum().sum()\n# print(nan_per_column)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.691826Z","iopub.execute_input":"2024-03-03T12:30:22.692092Z","iopub.status.idle":"2024-03-03T12:30:22.700041Z","shell.execute_reply.started":"2024-03-03T12:30:22.692071Z","shell.execute_reply":"2024-03-03T12:30:22.699258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nan_columns = nan_per_column[nan_per_column > 0].index.tolist()\n# print(\"Columns with NaN values:\", nan_columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.701300Z","iopub.execute_input":"2024-03-03T12:30:22.701570Z","iopub.status.idle":"2024-03-03T12:30:22.708853Z","shell.execute_reply.started":"2024-03-03T12:30:22.701549Z","shell.execute_reply":"2024-03-03T12:30:22.708160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n# imputer = SimpleImputer(strategy='mean')\n\n# numeric_columns = train.select_dtypes(include=['number']).columns\n\n# imputer.fit(train[numeric_columns])\n\n# train_imputed = imputer.transform(train[numeric_columns])\n\n# train[numeric_columns] = pd.DataFrame(train_imputed, columns=numeric_columns)\n\n# nan_per_column = train.isnull().sum().sum()\n# print(nan_per_column)\n\n# print(train.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.710216Z","iopub.execute_input":"2024-03-03T12:30:22.710529Z","iopub.status.idle":"2024-03-03T12:30:22.719828Z","shell.execute_reply.started":"2024-03-03T12:30:22.710501Z","shell.execute_reply":"2024-03-03T12:30:22.719055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical_columns = train.iloc[:, 12:].select_dtypes(include=np.number)\n# described = numerical_columns.describe()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.720827Z","iopub.execute_input":"2024-03-03T12:30:22.721379Z","iopub.status.idle":"2024-03-03T12:30:22.733598Z","shell.execute_reply.started":"2024-03-03T12:30:22.721347Z","shell.execute_reply":"2024-03-03T12:30:22.732752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(described)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.734792Z","iopub.execute_input":"2024-03-03T12:30:22.735119Z","iopub.status.idle":"2024-03-03T12:30:22.743129Z","shell.execute_reply.started":"2024-03-03T12:30:22.735096Z","shell.execute_reply":"2024-03-03T12:30:22.742269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(described.min().min())\n# print(described.max().max())","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.744190Z","iopub.execute_input":"2024-03-03T12:30:22.744459Z","iopub.status.idle":"2024-03-03T12:30:22.752513Z","shell.execute_reply.started":"2024-03-03T12:30:22.744437Z","shell.execute_reply":"2024-03-03T12:30:22.751766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n# def find_outliers_mean_std(df, col):\n#     mean_val = df[col].mean()\n#     std_val = df[col].std()\n#     outliers = df[df[col] > mean_val + 3 * std_val]\n#     return outliers\n\n# outliers = find_outliers_mean_std(numerical_columns, 'eeg_std_f506_10s')\n# print(outliers)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.753527Z","iopub.execute_input":"2024-03-03T12:30:22.753797Z","iopub.status.idle":"2024-03-03T12:30:22.762461Z","shell.execute_reply.started":"2024-03-03T12:30:22.753776Z","shell.execute_reply":"2024-03-03T12:30:22.761736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Somehow did not work well...\n\n# numerical_columns = train.iloc[:, 12:].select_dtypes(include='number')\n\n# has_negative_values = (numerical_columns < 0).any()\n\n# for col in numerical_columns.columns:\n#     if has_negative_values[col]:\n        \n#         train[col] = preprocessing.MinMaxScaler(feature_range=(-1, 1)).fit_transform(train[[col]])\n#     else:\n        \n#         train[col] = preprocessing.MinMaxScaler().fit_transform(train[[col]])","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.763704Z","iopub.execute_input":"2024-03-03T12:30:22.763989Z","iopub.status.idle":"2024-03-03T12:30:22.771620Z","shell.execute_reply.started":"2024-03-03T12:30:22.763968Z","shell.execute_reply":"2024-03-03T12:30:22.770805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.777540Z","iopub.execute_input":"2024-03-03T12:30:22.777819Z","iopub.status.idle":"2024-03-03T12:30:22.781555Z","shell.execute_reply.started":"2024-03-03T12:30:22.777797Z","shell.execute_reply":"2024-03-03T12:30:22.780680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FREE MEMORY\n# del all_eegs, data\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.782593Z","iopub.execute_input":"2024-03-03T12:30:22.782925Z","iopub.status.idle":"2024-03-03T12:30:22.791022Z","shell.execute_reply.started":"2024-03-03T12:30:22.782902Z","shell.execute_reply":"2024-03-03T12:30:22.790237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.792169Z","iopub.execute_input":"2024-03-03T12:30:22.792440Z","iopub.status.idle":"2024-03-03T12:30:22.800359Z","shell.execute_reply.started":"2024-03-03T12:30:22.792419Z","shell.execute_reply":"2024-03-03T12:30:22.799502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n# y = train.iloc[:, 11] \n# X = train.drop(columns=['target']) \n\n# # print(train.iloc[:, :12])\n# # print(X.iloc[:, :12])\n\n# # print(X)\n\n# smote = SMOTE(random_state=42)\n\n# X_resampled, y_resampled = smote.fit_resample(X, y)\n\n# train_resampled = pd.DataFrame(X_resampled, columns=X.columns)\n# train_resampled['target'] = y_resampled\n\n# unique_eeg_ids = set(train['eeg_id'])\n# unique_spec_ids = set(train['spec_id'])\n\n# for i, row in train_resampled.iterrows():\n#     # if i%100==0: print(i)\n#     while row['eeg_id'] in unique_eeg_ids:\n#         row['eeg_id'] += 1\n#     unique_eeg_ids.add(row['eeg_id'])\n\n#     while row['spec_id'] in unique_spec_ids:\n#         row['spec_id'] += 1\n#     unique_spec_ids.add(row['spec_id'])\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.801466Z","iopub.execute_input":"2024-03-03T12:30:22.801821Z","iopub.status.idle":"2024-03-03T12:30:22.810326Z","shell.execute_reply.started":"2024-03-03T12:30:22.801775Z","shell.execute_reply":"2024-03-03T12:30:22.809525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE\n\n# making less data (due to RAM issues in the first place)\n\n# class_counts = train_resampled['target'].value_counts()\n# min_class_count = 4000\n\n# train = pd.DataFrame()\n\n# for class_label in class_counts.index:\n#     class_subset = train_resampled[train_resampled['target'] == class_label]\n#     sampled_subset = class_subset.sample(n=min_class_count, random_state=42)\n#     train = pd.concat([train, sampled_subset], ignore_index=True)\n\n# train = train.sample(frac=1, random_state=42)\n\n# print(train.shape)\n# print(train['target'].value_counts())\n# BALANCED!","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.811308Z","iopub.execute_input":"2024-03-03T12:30:22.811559Z","iopub.status.idle":"2024-03-03T12:30:22.823340Z","shell.execute_reply.started":"2024-03-03T12:30:22.811538Z","shell.execute_reply":"2024-03-03T12:30:22.822635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_per_column = train.isnull().sum().sum()\nprint(nan_per_column)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:22.824321Z","iopub.execute_input":"2024-03-03T12:30:22.824544Z","iopub.status.idle":"2024-03-03T12:30:23.024240Z","shell.execute_reply.started":"2024-03-03T12:30:22.824525Z","shell.execute_reply":"2024-03-03T12:30:23.023320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_resampled","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:23.025427Z","iopub.execute_input":"2024-03-03T12:30:23.025789Z","iopub.status.idle":"2024-03-03T12:30:23.030951Z","shell.execute_reply.started":"2024-03-03T12:30:23.025757Z","shell.execute_reply":"2024-03-03T12:30:23.030100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:23.032135Z","iopub.execute_input":"2024-03-03T12:30:23.032408Z","iopub.status.idle":"2024-03-03T12:30:23.041000Z","shell.execute_reply.started":"2024-03-03T12:30:23.032385Z","shell.execute_reply":"2024-03-03T12:30:23.040150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n\n\n# FEATURE NAMES\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\n\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mdn_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_std_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_mdn_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_25%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_50%_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_75%_20s' for c in SPEC_COLS]\n\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:23.042107Z","iopub.execute_input":"2024-03-03T12:30:23.042502Z","iopub.status.idle":"2024-03-03T12:30:23.295967Z","shell.execute_reply.started":"2024-03-03T12:30:23.042472Z","shell.execute_reply":"2024-03-03T12:30:23.295093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARS = {'Seizure': 0, 'LPD': 1, 'GPD': 2, 'LRDA': 3, 'GRDA': 4, 'Other': 5}","metadata":{"execution":{"iopub.status.busy":"2024-03-03T12:30:23.297192Z","iopub.execute_input":"2024-03-03T12:30:23.297866Z","iopub.status.idle":"2024-03-03T12:30:23.302471Z","shell.execute_reply.started":"2024-03-03T12:30:23.297819Z","shell.execute_reply":"2024-03-03T12:30:23.301622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT USED ANYMORE, WE ALREADY HAVE THE 'OPTIMAL' HYPERPARAMETERS.\n\n# def optXGB(trial):\n    \n#     param = {\n#         'objective': 'multi:softprob',\n#         'num_class': len(TARS),\n#         'tree_method': 'gpu_hist',  \n#         'lambda': trial.suggest_loguniform('lambda', 1e-4, 10.0),\n#         'alpha': trial.suggest_loguniform('alpha', 1e-4, 10.0),\n#         'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.5, 0.6, 0.7, 0.8, 0.9, 1.0]),\n#         'subsample': trial.suggest_categorical('subsample', [0.6, 0.7, 0.8, 0.9, 1.0]),\n#         'learning_rate': trial.suggest_categorical('learning_rate', [0.01, 0.02, 0.05, 0.1]),\n#         'n_estimators': trial.suggest_categorical('n_estimators', [100, 500, 1000]),\n#         'max_depth': trial.suggest_categorical('max_depth', [5, 7, 9, 11, 13]),\n#         'min_child_weight': trial.suggest_int('min_child_weight', 1, 300)\n#     }\n\n#     gkf = GroupKFold(n_splits=5)\n#     cv_scores = []\n\n#     for train_index, valid_index in gkf.split(train, train.target, train.patient_id):\n#         X_train, X_valid = train.loc[train_index, FEATURES], train.loc[valid_index, FEATURES]\n#         y_train, y_valid = train.loc[train_index, 'target'].map(TARS), train.loc[valid_index, 'target'].map(TARS)\n\n#         model = xgb.XGBClassifier(**param)\n#         model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], verbose=False, early_stopping_rounds=10)\n#         preds = model.predict_proba(X_valid)\n#         cv_scores.append(log_loss(y_valid, preds))\n\n#     return np.mean(cv_scores)\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(optXGB, n_trials=10) \n\n# print('Number of finished trials:', len(study.trials))\n# print('Best trial:', study.best_trial.params)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-03T15:10:09.272267Z","iopub.execute_input":"2024-03-03T15:10:09.272746Z","iopub.status.idle":"2024-03-03T18:16:11.227716Z","shell.execute_reply.started":"2024-03-03T15:10:09.272704Z","shell.execute_reply":"2024-03-03T18:16:11.226832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LightGBM","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint(TARGETS)\n\nVersion = 1\n\nall_oof = []\nall_true = []\nTARS = {'Seizure': 0, 'LPD': 1, 'GPD': 2, 'LRDA': 3, 'GRDA': 4, 'Other': 5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):\n    \n    #print(len(X_valid.values[0]))\n    \n    print('#' * 25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#' * 25)\n    \n    model = lgb.LGBMClassifier(objective='multiclass', num_class=len(TARS))\n    \n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    callbacks = [\n        lgb.early_stopping(stopping_rounds=10),\n        lgb.log_evaluation(period=10) \n    ]\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              callbacks=callbacks)\n    \n    model.booster_.save_model(f'LightGBM_v{Version}_f{i}.txt')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    # print(\"ok\", train.loc[valid_index, TARGETS].values)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof_lgbm = all_oof\nall_true_lgbm = all_true","metadata":{"execution":{"iopub.status.busy":"2024-03-02T13:08:06.047246Z","iopub.execute_input":"2024-03-02T13:08:06.047590Z","iopub.status.idle":"2024-03-02T13:09:43.423316Z","shell.execute_reply.started":"2024-03-02T13:08:06.047565Z","shell.execute_reply":"2024-03-02T13:09:43.421874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"Version = 1\n\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train , train.target, train.patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = xgb.XGBClassifier(\n        objective='multi:softprob', \n        num_class=len(TARS),\n        learning_rate = 0.02,\n        reg_lambda = 0.09115581573097284,\n        alpha = 0.326474216065599,\n        colsample_bytree = 0.8,\n        subsample = 0.8,\n        n_estimators = 1000,\n        max_depth = 7,\n        min_child_weight = 68,\n        tree_method='gpu_hist'\n    )\n    \n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              verbose=True, \n              early_stopping_rounds=10)\n    model.save_model(f'XGB_v{Version}_f{i}.model')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof_xgb = np.concatenate(all_oof)\nall_true_xgb = np.concatenate(all_true)\n\nVersion = 1\n\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}","metadata":{"execution":{"iopub.status.busy":"2024-03-03T18:31:17.622665Z","iopub.execute_input":"2024-03-03T18:31:17.623330Z","iopub.status.idle":"2024-03-03T18:33:26.729276Z","shell.execute_reply.started":"2024-03-03T18:31:17.623300Z","shell.execute_reply":"2024-03-03T18:33:26.727751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CatBoost","metadata":{}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):\n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    \n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_v{Version}_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n    \nall_oof_cat = np.concatenate(all_oof)\nall_true_cat = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-03-03T10:26:00.768661Z","iopub.execute_input":"2024-03-03T10:26:00.769067Z","iopub.status.idle":"2024-03-03T10:26:19.800234Z","shell.execute_reply.started":"2024-03-03T10:26:00.769039Z","shell.execute_reply":"2024-03-03T10:26:19.798839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}