{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8015225,"sourceType":"datasetVersion","datasetId":4435105}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-11T05:46:16.108667Z","iopub.execute_input":"2024-04-11T05:46:16.109331Z","iopub.status.idle":"2024-04-11T05:46:16.119778Z","shell.execute_reply.started":"2024-04-11T05:46:16.109285Z","shell.execute_reply":"2024-04-11T05:46:16.118081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By looking at the data in time-series within the same eeg_id and considering it as different data when the label switches, the training data went from 17089 to 21490.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\nlabel_cols = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']\ndfs = []\nfor id, idf in tqdm(df.sort_values('eeg_label_offset_seconds').groupby('eeg_id')):\n    group_in_eegs = []\n    group_in_eeg = 0\n    this_labels = idf[label_cols].values[0]\n    for j, row, in idf.iterrows():\n        if not np.all(this_labels == row[label_cols]):\n            this_labels = row[label_cols]\n            group_in_eeg += 1\n        group_in_eegs.append(group_in_eeg)\n    idf['group_in_eeg'] = group_in_eegs\n    dfs.append(idf)\ndf = pd.concat(dfs)\ndf","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:20:12.849949Z","iopub.execute_input":"2024-04-11T06:20:12.850676Z","iopub.status.idle":"2024-04-11T06:21:49.894335Z","shell.execute_reply.started":"2024-04-11T06:20:12.850641Z","shell.execute_reply":"2024-04-11T06:21:49.893074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['vote_sum'] = df[['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']].sum(1)\ndf['eeg_path'] = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eeg/'+df.eeg_id.astype(str)+'.parquet'\ndf['spe_path'] = f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'+df.spectrogram_id.astype(str)+'.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:22:30.027630Z","iopub.execute_input":"2024-04-11T06:22:30.028573Z","iopub.status.idle":"2024-04-11T06:22:30.234236Z","shell.execute_reply.started":"2024-04-11T06:22:30.028537Z","shell.execute_reply":"2024-04-11T06:22:30.232765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fold by groupKfold, grouping_key='patient_id'\nfold_df = pd.read_csv('/kaggle/input/hms-data/train_unique_egg_v2.csv')[['patient_id', 'fold']].drop_duplicates('patient_id')\ndf = df.merge(fold_df, on='patient_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:24:39.356713Z","iopub.execute_input":"2024-04-11T06:24:39.357101Z","iopub.status.idle":"2024-04-11T06:24:39.512253Z","shell.execute_reply.started":"2024-04-11T06:24:39.357072Z","shell.execute_reply":"2024-04-11T06:24:39.511165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:24:42.914503Z","iopub.execute_input":"2024-04-11T06:24:42.914931Z","iopub.status.idle":"2024-04-11T06:24:42.942975Z","shell.execute_reply.started":"2024-04-11T06:24:42.914898Z","shell.execute_reply":"2024-04-11T06:24:42.941805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df.drop_duplicates('eeg_id'))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:26:34.560005Z","iopub.execute_input":"2024-04-11T06:26:34.560428Z","iopub.status.idle":"2024-04-11T06:26:34.576128Z","shell.execute_reply.started":"2024-04-11T06:26:34.560396Z","shell.execute_reply":"2024-04-11T06:26:34.574921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df.drop_duplicates(['eeg_id', 'group_in_eeg']))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:26:36.428468Z","iopub.execute_input":"2024-04-11T06:26:36.428864Z","iopub.status.idle":"2024-04-11T06:26:36.449458Z","shell.execute_reply.started":"2024-04-11T06:26:36.428832Z","shell.execute_reply":"2024-04-11T06:26:36.448305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By looking at the data in time-series within the same eeg_id and considering it as different data when the label switches, the training data went from 17089 to 21490.","metadata":{}},{"cell_type":"code","source":"df.to_csv('train_with_group_in_eeg.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T06:25:16.370305Z","iopub.execute_input":"2024-04-11T06:25:16.370936Z","iopub.status.idle":"2024-04-11T06:25:18.515115Z","shell.execute_reply.started":"2024-04-11T06:25:16.370905Z","shell.execute_reply":"2024-04-11T06:25:18.514099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}