{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HMS - HARMFUL BRAIN ACTIVITY CLASSIFICATION\n# Ezgi Gümüştekin-Emir Yorgun[](/https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/overview)","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\nVER = 1","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:09.188380Z","iopub.execute_input":"2025-06-14T19:48:09.188724Z","iopub.status.idle":"2025-06-14T19:48:11.426525Z","shell.execute_reply.started":"2025-06-14T19:48:09.188695Z","shell.execute_reply":"2025-06-14T19:48:11.425435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\nfile_paths = glob(os.path.join(BASE_PATH, '**', '*.parquet'), recursive=True)\ndf = pd.DataFrame({'path': file_paths})\ndf['test_type'] = df['path'].apply(\n    lambda p: os.path.basename(os.path.dirname(p)).split('_')[-1]\n)\ndf['id'] = df['path'].apply(\n    lambda p: os.path.splitext(os.path.basename(p))[0]\n)\ndf_eeg = pd.read_parquet(\n    os.path.join(BASE_PATH, 'train_eegs', '1000913311.parquet')\n)\ndf_eeg.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:11.427514Z","iopub.execute_input":"2025-06-14T19:48:11.427875Z","iopub.status.idle":"2025-06-14T19:48:51.184628Z","shell.execute_reply.started":"2025-06-14T19:48:11.427856Z","shell.execute_reply":"2025-06-14T19:48:51.183631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_channels = len(df_eeg.columns)\nn_channels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:51.186946Z","iopub.execute_input":"2025-06-14T19:48:51.187250Z","iopub.status.idle":"2025-06-14T19:48:51.194481Z","shell.execute_reply.started":"2025-06-14T19:48:51.187228Z","shell.execute_reply":"2025-06-14T19:48:51.193368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"csv_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\ndf = pd.read_csv(csv_path)\nTARGETS = df.columns[-6:]\nnum_rows, num_cols = df.shape\nprint(f\"Train.csv contains {num_rows} records across {num_cols} columns.\")\nprint(f\"Target columns: {TARGETS.tolist()}\")\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:51.195540Z","iopub.execute_input":"2025-06-14T19:48:51.195795Z","iopub.status.idle":"2025-06-14T19:48:51.483372Z","shell.execute_reply.started":"2025-06-14T19:48:51.195773Z","shell.execute_reply":"2025-06-14T19:48:51.482414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"segment_bounds = df.groupby('eeg_id')[['spectrogram_id', 'spectrogram_label_offset_seconds']].agg({\n    'spectrogram_id': 'first',\n    'spectrogram_label_offset_seconds': 'min'\n})\nsegment_bounds.columns = ['spec_id', 'min']\nend_times = df.groupby('eeg_id')[['spectrogram_label_offset_seconds']].agg('max')\nsegment_bounds['max'] = end_times\n\npatient_info = df.groupby('eeg_id')[['patient_id']].agg('first')\nsegment_bounds['patient_id'] = patient_info\n\ntarget_counts = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor label in TARGETS:\n    segment_bounds[label] = target_counts[label].values\n\nlabel_matrix = segment_bounds[TARGETS].values\nlabel_matrix = label_matrix / label_matrix.sum(axis=1, keepdims=True)\nsegment_bounds[TARGETS] = label_matrix\n\nexpert_labels = df.groupby('eeg_id')[['expert_consensus']].agg('first')\nsegment_bounds['target'] = expert_labels\n\ntrain = segment_bounds.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:51.484256Z","iopub.execute_input":"2025-06-14T19:48:51.484592Z","iopub.status.idle":"2025-06-14T19:48:51.601784Z","shell.execute_reply.started":"2025-06-14T19:48:51.484570Z","shell.execute_reply":"2025-06-14T19:48:51.600836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"READ_SPEC_FILES = False \nFEATURE_ENGINEER = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:51.603022Z","iopub.execute_input":"2025-06-14T19:48:51.603326Z","iopub.status.idle":"2025-06-14T19:48:51.607380Z","shell.execute_reply.started":"2025-06-14T19:48:51.603272Z","shell.execute_reply":"2025-06-14T19:48:51.606234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nSPECTROGRAM_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfile_list = os.listdir(SPECTROGRAM_DIR)\nprint(f'There are {len(file_list)} spectrogram parquets')\n\nif READ_SPEC_FILES:\n    spectrograms = {}\n    for idx, file_name in enumerate(file_list):\n        if idx % 100 == 0:\n            print(idx, ', ', end='')\n        spec_df = pd.read_parquet(f'{SPECTROGRAM_DIR}{file_name}')\n        spec_id = int(file_name.split('.')[0])\n        spectrograms[spec_id] = spec_df.iloc[:, 1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy', allow_pickle=True).item()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:48:51.608351Z","iopub.execute_input":"2025-06-14T19:48:51.608688Z","iopub.status.idle":"2025-06-14T19:49:43.035790Z","shell.execute_reply.started":"2025-06-14T19:48:51.608659Z","shell.execute_reply":"2025-06-14T19:49:43.035051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\nimport warnings\nwarnings.filterwarnings('ignore')\nSPEC_COLS = pd.read_parquet(f'{SPECTROGRAM_DIR }1000086677.parquet').columns[1:]\n\nFEATURES = [f'{col}_mean_15m' for col in SPEC_COLS]\nFEATURES += [f'{col}_min_15m' for col in SPEC_COLS]\nFEATURES += [f'{col}_mean_50s' for col in SPEC_COLS]\nFEATURES += [f'{col}_min_50s' for col in SPEC_COLS]\n\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ', end='')\n\nif FEATURE_ENGINEER:\n    feature_matrix = np.zeros((len(train), len(FEATURES)))\n    \n    for idx in range(len(train)):\n        if idx % 100 == 0:\n            print(idx, ', ', end='')\n        row_data = train.iloc[idx]\n        center_index = int((row_data['min'] + row_data['max']) // 4)\n\n        segment = spectrograms[row_data.spec_id][center_index:center_index + 450, :]\n        feature_vals = np.nanmean(segment, axis=0)\n        feature_matrix[idx, :400] = feature_vals\n        feature_vals = np.nanmin(segment, axis=0)\n        feature_matrix[idx, 400:800] = feature_vals\n        \n        short_segment = spectrograms[row_data.spec_id][center_index + 145:center_index + 170, :]\n        feature_vals = np.nanmean(short_segment, axis=0)\n        feature_matrix[idx, 800:1200] = feature_vals\n        feature_vals = np.nanmin(short_segment, axis=0)\n        feature_matrix[idx, 1200:1600] = feature_vals\n        \n    train[FEATURES] = feature_matrix\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\n\nprint()\nprint('New train shape:', train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:49:43.036628Z","iopub.execute_input":"2025-06-14T19:49:43.037040Z","iopub.status.idle":"2025-06-14T19:50:14.719353Z","shell.execute_reply.started":"2025-06-14T19:49:43.037016Z","shell.execute_reply":"2025-06-14T19:50:14.718233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:50:14.722498Z","iopub.execute_input":"2025-06-14T19:50:14.722766Z","iopub.status.idle":"2025-06-14T19:50:15.898058Z","shell.execute_reply.started":"2025-06-14T19:50:14.722746Z","shell.execute_reply":"2025-06-14T19:50:15.897162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_frequency_band_features(segment):\n    eeg_bands = {\n        'Delta': (0.5, 4),\n        'Theta': (4, 8),\n        'Alpha': (8, 12),\n        'Beta': (12, 30),\n        'Gamma': (30, 45)\n    }\n    band_features = []\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        bandpass_sos = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        filtered_signal = signal.sosfilt(bandpass_sos, segment)\n        band_features.extend([\n            np.nanmean(filtered_signal),  \n            np.nanstd(filtered_signal),    \n            np.nanmax(filtered_signal),    \n            np.nanmin(filtered_signal)     \n        ])\n    return band_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:50:15.898916Z","iopub.execute_input":"2025-06-14T19:50:15.899333Z","iopub.status.idle":"2025-06-14T19:50:15.906324Z","shell.execute_reply.started":"2025-06-14T19:50:15.899307Z","shell.execute_reply":"2025-06-14T19:50:15.905308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:50:15.907203Z","iopub.execute_input":"2025-06-14T19:50:15.907486Z","iopub.status.idle":"2025-06-14T19:50:15.960842Z","shell.execute_reply.started":"2025-06-14T19:50:15.907468Z","shell.execute_reply":"2025-06-14T19:50:15.959736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, GroupKFold\nprint('XGBoost version', xgb.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:50:15.962025Z","iopub.execute_input":"2025-06-14T19:50:15.962399Z","iopub.status.idle":"2025-06-14T19:50:16.277370Z","shell.execute_reply.started":"2025-06-14T19:50:15.962364Z","shell.execute_reply":"2025-06-14T19:50:16.276247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nimport xgboost as xgb\nimport gc\n\nall_oof   = []\nall_true  = []\nall_evals = []    \n\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\ngkf  = GroupKFold(n_splits=5)\n\nfor i, (train_index, valid_index) in enumerate(\n        gkf.split(train, train.target, train.patient_id)\n    ):\n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n\n    model = xgb.XGBClassifier(\n        objective='multi:softprob',\n        num_class=len(TARS),\n        learning_rate=0.1,\n        tree_method='hist', \n        eval_metric='mlogloss'\n    )\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        verbose=False,\n        early_stopping_rounds=10\n    )\n    \n    evals_result = model.evals_result()\n    val_logloss = evals_result['validation_0']['mlogloss']\n    all_evals.append(val_logloss)\n   \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n\n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n\nall_oof  = np.concatenate(all_oof, axis=0)\nall_true = np.concatenate(all_true, axis=0)\n\nplt.figure(figsize=(8, 6))\nfor fold_idx, logloss_curve in enumerate(all_evals):\n    plt.plot(\n        logloss_curve,\n        label=f'Fold {fold_idx+1}',\n        linewidth=1.5\n    )\n    \nplt.xlabel('Boosting Round', fontsize=12)\nplt.ylabel('Validation Log‐Loss', fontsize=12)\nplt.title('XGBoost Validation Log‐Loss vs. Boosting Round (5‐Fold)', fontsize=14)\nplt.legend(loc='upper right', fontsize=10)\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T19:55:34.853114Z","iopub.execute_input":"2025-06-14T19:55:34.853455Z","iopub.status.idle":"2025-06-14T20:16:57.912429Z","shell.execute_reply.started":"2025-06-14T19:55:34.853431Z","shell.execute_reply":"2025-06-14T20:16:57.911326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TOP_K = 30\nimportances = model.feature_importances_\nall_features = train.columns\nranking_indices = np.argsort(importances)\n\nplt.figure(figsize=(10, 8))\ntop_indices = ranking_indices[-TOP_K:]\nplt.barh(np.arange(TOP_K), importances[top_indices], align='center')\nplt.yticks(np.arange(TOP_K), all_features[top_indices])\nplt.title(f'Top {TOP_K} Feature Importances')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:46.297195Z","iopub.execute_input":"2025-06-14T20:19:46.297642Z","iopub.status.idle":"2025-06-14T20:19:46.731610Z","shell.execute_reply.started":"2025-06-14T20:19:46.297617Z","shell.execute_reply":"2025-06-14T20:19:46.730557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:46.733028Z","iopub.execute_input":"2025-06-14T20:19:46.733323Z","iopub.status.idle":"2025-06-14T20:19:46.760358Z","shell.execute_reply.started":"2025-06-14T20:19:46.733291Z","shell.execute_reply":"2025-06-14T20:19:46.759496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s = 853520\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\nspec = pd.read_parquet(f'{PATH2}{s}.parquet')\nspec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:46.761519Z","iopub.execute_input":"2025-06-14T20:19:46.762262Z","iopub.status.idle":"2025-06-14T20:19:46.849002Z","shell.execute_reply.started":"2025-06-14T20:19:46.762232Z","shell.execute_reply":"2025-06-14T20:19:46.848045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES))) \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n\ntest[FEATURES] = data\nprint('New test shape',test.shape)\nprint(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:46.850691Z","iopub.execute_input":"2025-06-14T20:19:46.850972Z","iopub.status.idle":"2025-06-14T20:19:47.677530Z","shell.execute_reply.started":"2025-06-14T20:19:46.850952Z","shell.execute_reply":"2025-06-14T20:19:47.676535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = []\nfor i in range(5):\n    print(i, ', ', end='')\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    pred = model.predict_proba(test[FEATURES])\n    preds.append(pred)\n\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)\nprint(pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:47.678320Z","iopub.execute_input":"2025-06-14T20:19:47.678616Z","iopub.status.idle":"2025-06-14T20:19:48.665920Z","shell.execute_reply.started":"2025-06-14T20:19:47.678586Z","shell.execute_reply":"2025-06-14T20:19:48.665021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()\nprint(sub)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:48.666973Z","iopub.execute_input":"2025-06-14T20:19:48.667261Z","iopub.status.idle":"2025-06-14T20:19:48.687670Z","shell.execute_reply.started":"2025-06-14T20:19:48.667242Z","shell.execute_reply":"2025-06-14T20:19:48.686698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.iloc[:,-6:].sum(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T20:19:48.688932Z","iopub.execute_input":"2025-06-14T20:19:48.689237Z","iopub.status.idle":"2025-06-14T20:19:48.703106Z","shell.execute_reply.started":"2025-06-14T20:19:48.689210Z","shell.execute_reply":"2025-06-14T20:19:48.702174Z"}},"outputs":[],"execution_count":null}]}