{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc\n\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\nVER =1","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:20.745167Z","iopub.execute_input":"2024-02-03T01:16:20.745646Z","iopub.status.idle":"2024-02-03T01:16:20.751448Z","shell.execute_reply.started":"2024-02-03T01:16:20.745601Z","shell.execute_reply":"2024-02-03T01:16:20.750686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the reading of one parquet for understanding\n\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\n\ndf","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:20.812741Z","iopub.execute_input":"2024-02-03T01:16:20.813208Z","iopub.status.idle":"2024-02-03T01:16:21.464242Z","shell.execute_reply.started":"2024-02-03T01:16:20.813168Z","shell.execute_reply":"2024-02-03T01:16:21.463047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['path'][0]","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.466103Z","iopub.execute_input":"2024-02-03T01:16:21.466424Z","iopub.status.idle":"2024-02-03T01:16:21.474070Z","shell.execute_reply.started":"2024-02-03T01:16:21.466396Z","shell.execute_reply":"2024-02-03T01:16:21.472887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['test_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.475312Z","iopub.execute_input":"2024-02-03T01:16:21.475626Z","iopub.status.idle":"2024-02-03T01:16:21.490267Z","shell.execute_reply.started":"2024-02-03T01:16:21.475586Z","shell.execute_reply":"2024-02-03T01:16:21.489092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape)\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.492997Z","iopub.execute_input":"2024-02-03T01:16:21.493316Z","iopub.status.idle":"2024-02-03T01:16:21.831466Z","shell.execute_reply.started":"2024-02-03T01:16:21.493288Z","shell.execute_reply":"2024-02-03T01:16:21.830662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n{'spectrogram_label_offset_seconds':max})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1, keepdims = True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlap eeg_idd shape:', train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.833171Z","iopub.execute_input":"2024-02-03T01:16:21.833469Z","iopub.status.idle":"2024-02-03T01:16:21.941188Z","shell.execute_reply.started":"2024-02-03T01:16:21.833443Z","shell.execute_reply":"2024-02-03T01:16:21.940047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_SPEC_FILES = False # IF READ_SPEC_FILES is False, the code readds the combined file instead of individual files.\nFEATURE_ENGINEER = True\nREAD_EEG_SPEC_FILES = False","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.942799Z","iopub.execute_input":"2024-02-03T01:16:21.943247Z","iopub.status.idle":"2024-02-03T01:16:21.950077Z","shell.execute_reply.started":"2024-02-03T01:16:21.943203Z","shell.execute_reply":"2024-02-03T01:16:21.948439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:16:21.952243Z","iopub.execute_input":"2024-02-03T01:16:21.959451Z","iopub.status.idle":"2024-02-03T01:17:49.060952Z","shell.execute_reply.started":"2024-02-03T01:16:21.959399Z","shell.execute_reply":"2024-02-03T01:17:49.059932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# The code generates features from the spectrogram data for use in a model \n# The features are derived by calculating the mean and minimum values over time for each of the 400 spectrogram frequencies.\n# Two types of windows are used for these calculations:\n# A 10-minute window (_mean_10m, _min_10m).\n# A 20-second window (_mean_20s, _min_20s).\n# This process results in 1600 features (400 features × 4 calculations) for each EEG ID.\n\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\n\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\n# A data matrix data is initialized to store the new features for each eeg_id in the train DataFrame.\n# For each row in train, the code calculates the mean and minimum values within the specified 10-minute and 20-second windows.\n# These calculated values are then stored in the data matrix.\n# Finally, the matrix is added to the train DataFrame as new columns.\n\nif FEATURE_ENGINEER:\n    data = np.zeros((len(train),len(FEATURES)))\n    for k in range(len(train)):\n        if k%100==0: print(k,', ',end='')\n        row = train.iloc[k]\n        r = int( (row['min'] + row['max'])//4 ) \n        \n        # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,:400] = x\n        x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,400:800] = x\n        \n        # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,800:1200] = x\n        x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,1200:1600] = x\n        \n    train[FEATURES] = data\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\nprint()\nprint('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:17:49.062544Z","iopub.execute_input":"2024-02-03T01:17:49.062929Z","iopub.status.idle":"2024-02-03T01:18:20.625529Z","shell.execute_reply.started":"2024-02-03T01:17:49.062896Z","shell.execute_reply":"2024-02-03T01:18:20.624284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:18:20.627592Z","iopub.execute_input":"2024-02-03T01:18:20.628427Z","iopub.status.idle":"2024-02-03T01:18:22.159472Z","shell.execute_reply.started":"2024-02-03T01:18:20.628376Z","shell.execute_reply":"2024-02-03T01:18:22.158067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_frequency_band_features(segment):\n    # Define EEG frequency bands\n    eeg_bands = {'Delta': (0.5, 4), 'Theta': (4, 8), 'Alpha': (8, 12), 'Beta': (12, 30), 'Gamma': (30, 45)}\n    \n    band_features = []\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        # Filter signal for the specific band\n        band_pass_filter = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        filtered = signal.sosfilt(band_pass_filter, segment)\n        # Extract features like mean, standard deviation, etc.\n        band_features.extend([np.nanmean(filtered), np.nanstd(filtered), np.nanmax(filtered), np.nanmin(filtered)])\n    \n    return band_features","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:18:22.164744Z","iopub.execute_input":"2024-02-03T01:18:22.165178Z","iopub.status.idle":"2024-02-03T01:18:22.173776Z","shell.execute_reply.started":"2024-02-03T01:18:22.165136Z","shell.execute_reply":"2024-02-03T01:18:22.172652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:18:22.175591Z","iopub.execute_input":"2024-02-03T01:18:22.176037Z","iopub.status.idle":"2024-02-03T01:18:22.235039Z","shell.execute_reply.started":"2024-02-03T01:18:22.175995Z","shell.execute_reply":"2024-02-03T01:18:22.233947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\n\nprint('XGBoost version', xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:18:22.236423Z","iopub.execute_input":"2024-02-03T01:18:22.237540Z","iopub.status.idle":"2024-02-03T01:18:22.422655Z","shell.execute_reply.started":"2024-02-03T01:18:22.237494Z","shell.execute_reply":"2024-02-03T01:18:22.421401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train , train .target, train .patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = xgb.XGBClassifier(\n        objective='multi:softprob', \n        num_class=len(TARS),\n        learning_rate = 0.1, \n                      \n#         tree_method='gpu_hist',  #skip GPU acceleration\n    )\n    \n    # Prepare training and validation data\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              verbose=True, \n              early_stopping_rounds=10)\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:18:22.424068Z","iopub.execute_input":"2024-02-03T01:18:22.424424Z","iopub.status.idle":"2024-02-03T01:39:53.199068Z","shell.execute_reply.started":"2024-02-03T01:18:22.424393Z","shell.execute_reply":"2024-02-03T01:39:53.197492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOP = 30\n\n# Assuming 'model' is your trained model\nfeature_importance = model.feature_importances_\n\n# Get the feature names from 'train'\nfeature_names = train.columns\n\n# Sort the feature importances and get the indices of the sorted array\nsorted_idx = np.argsort(feature_importance)\n\n# Plot only the top 'TOP' features\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], feature_names[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:39:53.202138Z","iopub.execute_input":"2024-02-03T01:39:53.202572Z","iopub.status.idle":"2024-02-03T01:39:53.872592Z","shell.execute_reply.started":"2024-02-03T01:39:53.202529Z","shell.execute_reply":"2024-02-03T01:39:53.871376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/kaggle-kl-div')\nfrom kaggle_kl_div import score\n\noof = pd.DataFrame(all_oof.copy())\noof['id'] = np.arange(len(oof))\n\ntrue = pd.DataFrame(all_true.copy())\ntrue['id'] = np.arange(len(true))\n\ncv = score(solution=true, submission=oof, row_id_column_name='id')\nprint('CV Score KL-Div for CatBoost =',cv)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:39:53.874451Z","iopub.execute_input":"2024-02-03T01:39:53.874829Z","iopub.status.idle":"2024-02-03T01:39:53.970594Z","shell.execute_reply.started":"2024-02-03T01:39:53.874794Z","shell.execute_reply":"2024-02-03T01:39:53.969439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:39:53.971977Z","iopub.execute_input":"2024-02-03T01:39:53.972342Z","iopub.status.idle":"2024-02-03T01:39:53.990763Z","shell.execute_reply.started":"2024-02-03T01:39:53.972312Z","shell.execute_reply":"2024-02-03T01:39:53.989989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n\n    \n    \n    \ntest[FEATURES] = data\nprint('New test shape',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:43:18.639921Z","iopub.execute_input":"2024-02-03T01:43:18.640839Z","iopub.status.idle":"2024-02-03T01:43:19.419899Z","shell.execute_reply.started":"2024-02-03T01:43:18.640786Z","shell.execute_reply":"2024-02-03T01:43:19.418511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER XGBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    # Load the XGBoost model\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    \n    # Make predictions\n    pred = model.predict_proba(test[FEATURES])\n    preds.append(pred)\n\n# Average the predictions from each fold\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:43:22.431259Z","iopub.execute_input":"2024-02-03T01:43:22.431931Z","iopub.status.idle":"2024-02-03T01:43:24.277981Z","shell.execute_reply.started":"2024-02-03T01:43:22.431894Z","shell.execute_reply":"2024-02-03T01:43:24.276775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:43:27.337878Z","iopub.execute_input":"2024-02-03T01:43:27.338640Z","iopub.status.idle":"2024-02-03T01:43:27.363164Z","shell.execute_reply.started":"2024-02-03T01:43:27.338589Z","shell.execute_reply":"2024-02-03T01:43:27.362376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T01:43:35.528863Z","iopub.execute_input":"2024-02-03T01:43:35.529951Z","iopub.status.idle":"2024-02-03T01:43:35.542083Z","shell.execute_reply.started":"2024-02-03T01:43:35.529891Z","shell.execute_reply":"2024-02-03T01:43:35.540776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}