{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\nfrom pathlib import Path\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification/')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-19T08:02:54.22825Z","iopub.execute_input":"2024-02-19T08:02:54.228608Z","iopub.status.idle":"2024-02-19T08:02:54.233403Z","shell.execute_reply.started":"2024-02-19T08:02:54.228574Z","shell.execute_reply":"2024-02-19T08:02:54.232465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Debug setting, to only look at 5% of the data initially\nDEBUG = True\n\n# Reading in the data\ndf = pd.read_csv(path/'train.csv')\nif DEBUG:\n#     df = df.iloc[int(len(df)*0.5)]\n    df = df.iloc[:int(len(df) * 0.05)]\n\ntest_df = pd.read_csv(path/'test.csv')\ntrain_spect = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')\nTARGETS = df.columns[-6:]\n\n# Approach I used here was from https://www.kaggle.com/code/yorkyong/exploring-eeg-a-beginner-s-guide\n\n# Here we take one EEG ID and match it to a spectogram, looking for the first available spectogram in the series\ntrain = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\n# Here we find the last point in the EEG to be matched by a spectogram\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\n# Adding the patient ID to the EEG\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first') # The code adds the patient_id for each eeg_id to the train DataFrame. This links each EEG segment to a specific patient.\ntrain['patient_id'] = tmp\n\n# Adding the targets to the dataframe\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \n# Now normalising the data so that all the targets add up to 1\ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1, keepdims=True)\n\n# Adding the expert consensus\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first') # For each eeg_id, the code includes the expert_consensus on the EEG segment's classification.\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.234624Z","iopub.execute_input":"2024-02-19T08:02:54.234989Z","iopub.status.idle":"2024-02-19T08:02:54.488755Z","shell.execute_reply.started":"2024-02-19T08:02:54.234964Z","shell.execute_reply":"2024-02-19T08:02:54.487832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nDEBUG = True\n\n# Below code worked for the whole set, trying to refactor it so it only looks at the debugging subset\n\n# # Reading in the spectogram files\n# path_spect = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n# files = os.listdir(path_spect)\n# print(f'There are {len(files)} spectrogram parquets')\n\n# spectograms ={}\n# for i, f in enumerate(files):\n#     if i%100==0: print(i,', ',end='')\n#     tmp = pd.read_parquet(f'{path_spect}{f}')\n#     name = int(f.split('.')[0])\n#     spectograms[name] = tmp.iloc[:,1:].values\n\n# Refactored code:\npath_spect = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(path_spect)\n\nif DEBUG:\n    spec_ids = set(train['spec_id'].astype(str))\n    relevant_files = [f for f in files if f.split('.')[0] in spec_ids]\n    spectograms = {}\n    for i, f in enumerate(relevant_files):  # Iterate only over filtered files\n        if i % 100 == 0:\n            print(i, ', ', end='')\nelse:\n    spectograms ={}\n    for i, f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{path_spect}{f}')\n        name = int(f.split('.')[0])\n        spectograms[name] = tmp.iloc[:,1:].values\n        \n\n\n# # Filter files to include only those whose IDs are in spec_ids\n# # Assuming filenames are just the spec_id with an extension, e.g., \"123456.parquet\"\n# relevant_files = [f for f in files if f.split('.')[0] in spec_ids]\n\n# spectograms = {}\n# for i, f in enumerate(relevant_files):  # Iterate only over filtered files\n#     if i % 100 == 0:\n#         print(i, ', ', end='')\n\n#     tmp = pd.read_parquet(f'{path_spect}{f}')\n#     name = int(f.split('.')[0])  # Extract the spectrogram ID from the filename\n#     spectograms[name] = tmp.iloc[:,1:].values\n        ","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.489971Z","iopub.execute_input":"2024-02-19T08:02:54.490282Z","iopub.status.idle":"2024-02-19T08:02:54.508493Z","shell.execute_reply.started":"2024-02-19T08:02:54.490255Z","shell.execute_reply":"2024-02-19T08:02:54.507591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Making features, here we are taking the mean and minumum features for 10 and 20 second intervals\nFEATURE_ENGINEER = True\n\nSPEC_COLS = pd.read_parquet(f'{path_spect}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\nif FEATURE_ENGINEER:\n    data = np.zeros((len(train),len(FEATURES)))\n    for k in range(len(train)):\n        if k%100==0: print(k,', ',end='')\n        row = train.iloc[k]\n        r = int( (row['min'] + row['max'])//4 ) \n        \n        # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,:400] = x\n        x = np.nanmin(spectograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,400:800] = x\n        \n        # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,800:1200] = x\n        x = np.nanmin(spectograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,1200:1600] = x\n\n    train[FEATURES] = data\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\nprint()\nprint('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.51165Z","iopub.execute_input":"2024-02-19T08:02:54.512011Z","iopub.status.idle":"2024-02-19T08:02:54.564417Z","shell.execute_reply.started":"2024-02-19T08:02:54.511987Z","shell.execute_reply":"2024-02-19T08:02:54.563564Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_COLS\n","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:06:38.441898Z","iopub.execute_input":"2024-02-19T08:06:38.442542Z","iopub.status.idle":"2024-02-19T08:06:38.448309Z","shell.execute_reply.started":"2024-02-19T08:06:38.442513Z","shell.execute_reply":"2024-02-19T08:06:38.447508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA\ndef extract_frequency_band_features(segment):\n    # Define EEG frequency bands\n    eeg_bands = {'Delta': (0.5, 4), 'Theta': (4, 8), 'Alpha': (8, 12), 'Beta': (12, 30), 'Gamma': (30, 45)}\n    \n    band_features = []\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        # Filter signal for the specific band\n        band_pass_filter = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        filtered = signal.sosfilt(band_pass_filter, segment)\n        # Extract features like mean, standard deviation, etc.\n        band_features.extend([np.nanmean(filtered), np.nanstd(filtered), np.nanmax(filtered), np.nanmin(filtered)])\n    \n    return band_features\n","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.57248Z","iopub.execute_input":"2024-02-19T08:02:54.572791Z","iopub.status.idle":"2024-02-19T08:02:54.58099Z","shell.execute_reply.started":"2024-02-19T08:02:54.572745Z","shell.execute_reply":"2024-02-19T08:02:54.580139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\n\nprint('XGBoost version', xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.581985Z","iopub.execute_input":"2024-02-19T08:02:54.582213Z","iopub.status.idle":"2024-02-19T08:02:54.598578Z","shell.execute_reply.started":"2024-02-19T08:02:54.582193Z","shell.execute_reply":"2024-02-19T08:02:54.597733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.59959Z","iopub.execute_input":"2024-02-19T08:02:54.599879Z","iopub.status.idle":"2024-02-19T08:02:54.622221Z","shell.execute_reply.started":"2024-02-19T08:02:54.599857Z","shell.execute_reply":"2024-02-19T08:02:54.621287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nVER = 1\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\n# train = train.reset_index() # To remove on next run\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train , train .target, train .patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = xgb.XGBClassifier(\n        objective='multi:softprob', \n        num_class=len(TARS),\n        learning_rate = 0.1, \n                      \n        tree_method='gpu_hist',  #skip GPU acceleration\n    )\n    \n    # Prepare training and validation data\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              verbose=True, \n              early_stopping_rounds=10)\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T08:02:54.623386Z","iopub.execute_input":"2024-02-19T08:02:54.62372Z","iopub.status.idle":"2024-02-19T08:02:54.731454Z","shell.execute_reply.started":"2024-02-19T08:02:54.623687Z","shell.execute_reply":"2024-02-19T08:02:54.730574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}