{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":161856283,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nVER = 2","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:01.142299Z","iopub.execute_input":"2024-02-18T06:50:01.142675Z","iopub.status.idle":"2024-02-18T06:50:02.049402Z","shell.execute_reply.started":"2024-02-18T06:50:01.142647Z","shell.execute_reply":"2024-02-18T06:50:02.048346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf['expert_consensus'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:02.051318Z","iopub.execute_input":"2024-02-18T06:50:02.052200Z","iopub.status.idle":"2024-02-18T06:50:02.698330Z","shell.execute_reply.started":"2024-02-18T06:50:02.052168Z","shell.execute_reply":"2024-02-18T06:50:02.697412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = list(df.columns[-6:])\nTARGETS_GROUP2 = ['other_vote']\nTARGETS_GROUP1 = list(set(list(df.columns[-6:])) - set(TARGETS_GROUP2))\ndf['all_target_group1'] = df.apply(lambda row: sum([row[x] for x in TARGETS_GROUP1]) , axis=1)\ndf['all_target_group2'] = df.apply(lambda row: sum([row[x] for x in TARGETS_GROUP2]) , axis=1)\nTARGETS = ['all_target_group1','all_target_group2']\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-18T06:50:02.699472Z","iopub.execute_input":"2024-02-18T06:50:02.699784Z","iopub.status.idle":"2024-02-18T06:50:07.615625Z","shell.execute_reply.started":"2024-02-18T06:50:02.699757Z","shell.execute_reply":"2024-02-18T06:50:07.614644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Non-Overlapping Eeg Id Train Data\nThe competition data description says that test data does not have multiple crops from the same `eeg_id`. Therefore we will train and validate using only 1 crop per `eeg_id`. There is a discussion about this [here][1].\n\n[1]: https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/discussion/467021","metadata":{}},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:07.616936Z","iopub.execute_input":"2024-02-18T06:50:07.617238Z","iopub.status.idle":"2024-02-18T06:50:07.690654Z","shell.execute_reply.started":"2024-02-18T06:50:07.617212Z","shell.execute_reply":"2024-02-18T06:50:07.689461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['org_target'] = tmp\n\nTARGETS_GROUP1_NAMES = [x.lower().replace(\"_vote\",\"\").strip() for x in TARGETS_GROUP1]\ntrain['target'] = train.apply(lambda x: \"Group1\" if  x['org_target'].lower() in  TARGETS_GROUP1_NAMES else \"Group2\", axis=1)\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:07.693238Z","iopub.execute_input":"2024-02-18T06:50:07.693564Z","iopub.status.idle":"2024-02-18T06:50:07.921477Z","shell.execute_reply.started":"2024-02-18T06:50:07.693537Z","shell.execute_reply":"2024-02-18T06:50:07.920534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['org_target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:07.922737Z","iopub.execute_input":"2024-02-18T06:50:07.923041Z","iopub.status.idle":"2024-02-18T06:50:08.206961Z","shell.execute_reply.started":"2024-02-18T06:50:07.923015Z","shell.execute_reply":"2024-02-18T06:50:08.206000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:08.208044Z","iopub.execute_input":"2024-02-18T06:50:08.208342Z","iopub.status.idle":"2024-02-18T06:50:08.460104Z","shell.execute_reply.started":"2024-02-18T06:50:08.208316Z","shell.execute_reply":"2024-02-18T06:50:08.459136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nIn this section, we create features for our CatBoost model. \n\nFirst we need to read in all 11k train spectrogram files. Reading thousands of files takes 11 minutes with Pandas. Instead, we can read 1 file from my [Kaggle dataset here][1] which contains all the 11k spectrograms in less than 1 minute! To use my [Kaggle dataset][1], set variable `READ_SPEC_FILES = False`. Don't forget to upvote this helpful [dataset][1] :-)\n\nNext we need to engineer features for our CatBoost model. In this notebook, we just take the mean (over time) of each of the 400 spectrogram frequencies (using middle 10 minutes). This produces 400 features (per each unique eeg id). We can improve CV and LB score by engineering new features (and/or tuning CatBoost).\n\nUPDATE: Version 2 creates features from `means` and `mins`. And version 2 uses `10 minute windows` and `20 second windows`.\n\n[1]: https://www.kaggle.com/datasets/cdeotte/brain-spectrograms","metadata":{}},{"cell_type":"code","source":"READ_SPEC_FILES = False\nREAD_EEG_SPEC_FILES = False","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:08.461374Z","iopub.execute_input":"2024-02-18T06:50:08.461740Z","iopub.status.idle":"2024-02-18T06:50:08.466013Z","shell.execute_reply.started":"2024-02-18T06:50:08.461712Z","shell.execute_reply":"2024-02-18T06:50:08.465148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES: \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:50:08.466867Z","iopub.execute_input":"2024-02-18T06:50:08.467161Z","iopub.status.idle":"2024-02-18T06:51:06.516913Z","shell.execute_reply.started":"2024-02-18T06:50:08.467126Z","shell.execute_reply":"2024-02-18T06:51:06.515966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL EEG SPECTROGRAMS\nif READ_EEG_SPEC_FILES:\n    all_eegs = {}\n    for i,e in enumerate(train.eeg_id.values):\n        if i%100==0: print(i,', ',end='')\n        x = np.load(f'/kaggle/input/brain-eeg-spectrograms/EEG_Spectrograms/{e}.npy')\n        all_eegs[e] = x\nelse:\n    all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:51:06.518375Z","iopub.execute_input":"2024-02-18T06:51:06.518691Z","iopub.status.idle":"2024-02-18T06:52:16.102659Z","shell.execute_reply.started":"2024-02-18T06:51:06.518662Z","shell.execute_reply":"2024-02-18T06:52:16.101647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\n\n# FEATURE NAMES\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\ndata = np.zeros((len(train),len(FEATURES)))\nfor k in tqdm(range(len(train))):\n    row = train.iloc[k]\n    r = int( (row['min'] + row['max'])//4 ) \n#     r = int( row['max']//2 ) \n\n\n    # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,:400] = x\n    x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,1200:1600] = x\n\n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,1600:2112] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2112:2624] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2624:3136] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,3136:3648] = x\n\ntrain[FEATURES] = data\nprint(); print('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:52:16.104190Z","iopub.execute_input":"2024-02-18T06:52:16.104594Z","iopub.status.idle":"2024-02-18T06:53:10.748729Z","shell.execute_reply.started":"2024-02-18T06:52:16.104556Z","shell.execute_reply":"2024-02-18T06:53:10.747757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:53:10.750042Z","iopub.execute_input":"2024-02-18T06:53:10.750343Z","iopub.status.idle":"2024-02-18T06:53:10.811601Z","shell.execute_reply.started":"2024-02-18T06:53:10.750316Z","shell.execute_reply":"2024-02-18T06:53:10.810546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train CatBoost\nWe use the default settings for CatBoost which are pretty good. We can tune CatBoost manually to improve CV and LB score. Note that CatBoost will automatically use both Kaggle T4 GPUs (when we add parameter `task_type='GPU'`)  for super fast training!","metadata":{}},{"cell_type":"code","source":"import catboost as cat, gc\nfrom catboost import CatBoostClassifier, Pool\nprint('CatBoost version',cat.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:53:10.813176Z","iopub.execute_input":"2024-02-18T06:53:10.814056Z","iopub.status.idle":"2024-02-18T06:53:12.043533Z","shell.execute_reply.started":"2024-02-18T06:53:10.814014Z","shell.execute_reply":"2024-02-18T06:53:12.042520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[TARGETS]","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:53:12.047642Z","iopub.execute_input":"2024-02-18T06:53:12.047945Z","iopub.status.idle":"2024-02-18T06:53:12.060924Z","shell.execute_reply.started":"2024-02-18T06:53:12.047918Z","shell.execute_reply":"2024-02-18T06:53:12.059730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold, StratifiedGroupKFold\n\nall_oof = []\nall_true = []\nTARS = {'Group1':0, 'Group2':1}\n\n# gkf = GroupKFold(n_splits=5)\nall_index = []\n# for i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):   \nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=42)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(train, train['target'], groups=train['patient_id'])):\n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_v{VER}_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    all_index.extend(valid_index)\n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:53:12.062659Z","iopub.execute_input":"2024-02-18T06:53:12.062955Z","iopub.status.idle":"2024-02-18T06:56:41.126170Z","shell.execute_reply.started":"2024-02-18T06:53:12.062930Z","shell.execute_reply":"2024-02-18T06:56:41.125037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf = df[df['expert_consensus'] != 'Other']\nTARGETS = list(df.columns[-6:])\nTARGETS = list(set(list(df.columns[-6:])) - set(['other_vote']))\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:56:41.127730Z","iopub.execute_input":"2024-02-18T06:56:41.128389Z","iopub.status.idle":"2024-02-18T06:56:41.573705Z","shell.execute_reply.started":"2024-02-18T06:56:41.128353Z","shell.execute_reply":"2024-02-18T06:56:41.572771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:56:41.575926Z","iopub.execute_input":"2024-02-18T06:56:41.576648Z","iopub.status.idle":"2024-02-18T06:56:41.646987Z","shell.execute_reply.started":"2024-02-18T06:56:41.576610Z","shell.execute_reply":"2024-02-18T06:56:41.646049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:56:41.648625Z","iopub.execute_input":"2024-02-18T06:56:41.649314Z","iopub.status.idle":"2024-02-18T06:56:41.696536Z","shell.execute_reply.started":"2024-02-18T06:56:41.649277Z","shell.execute_reply":"2024-02-18T06:56:41.695714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:56:41.697995Z","iopub.execute_input":"2024-02-18T06:56:41.698647Z","iopub.status.idle":"2024-02-18T06:56:41.964617Z","shell.execute_reply.started":"2024-02-18T06:56:41.698611Z","shell.execute_reply":"2024-02-18T06:56:41.963687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\n\n# FEATURE NAMES\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\ndata = np.zeros((len(train),len(FEATURES)))\nfor k in tqdm(range(len(train))):\n    row = train.iloc[k]\n    r = int( (row['min'] + row['max'])//4 ) \n#     r = int( row['max']//2 ) \n\n    \n\n    # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,:400] = x\n    x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,1200:1600] = x\n\n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,1600:2112] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2112:2624] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2624:3136] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,3136:3648] = x\n\ntrain[FEATURES] = data\nprint(); print('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:56:41.965964Z","iopub.execute_input":"2024-02-18T06:56:41.966263Z","iopub.status.idle":"2024-02-18T06:57:18.092749Z","shell.execute_reply.started":"2024-02-18T06:56:41.966236Z","shell.execute_reply":"2024-02-18T06:57:18.091805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\n\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4}\n\n# gkf = GroupKFold(n_splits=5)\nall_index = []\n# for i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):   \nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=42)\nfor i, (train_index, valid_index) in enumerate(sgkf.split(train, train['target'], groups=train['patient_id'])):\n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_multi_v{VER}_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    all_index.extend(valid_index)\n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T06:57:18.094129Z","iopub.execute_input":"2024-02-18T06:57:18.094946Z","iopub.status.idle":"2024-02-18T07:03:35.931860Z","shell.execute_reply.started":"2024-02-18T06:57:18.094906Z","shell.execute_reply":"2024-02-18T07:03:35.930839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train; gc.collect()\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:35.934335Z","iopub.execute_input":"2024-02-18T07:03:35.935287Z","iopub.status.idle":"2024-02-18T07:03:35.960821Z","shell.execute_reply.started":"2024-02-18T07:03:35.935250Z","shell.execute_reply":"2024-02-18T07:03:35.959994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n# test = test[['spectrogram_id', 'eeg_id', 'patient_id']].groupby('eeg_id').agg({'spectrogram_id': 'first', 'patient_id':'first'}).sample(2700).reset_index()\n# test","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:35.961853Z","iopub.execute_input":"2024-02-18T07:03:35.962757Z","iopub.status.idle":"2024-02-18T07:03:35.966863Z","shell.execute_reply.started":"2024-02-18T07:03:35.962723Z","shell.execute_reply":"2024-02-18T07:03:35.965642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt, librosa\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:35.968423Z","iopub.execute_input":"2024-02-18T07:03:35.968945Z","iopub.status.idle":"2024-02-18T07:03:36.112873Z","shell.execute_reply.started":"2024-02-18T07:03:35.968911Z","shell.execute_reply":"2024-02-18T07:03:36.111851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATE ALL EEG SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 0\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(tqdm(EEG_IDS2)):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:36.114226Z","iopub.execute_input":"2024-02-18T07:03:36.115240Z","iopub.status.idle":"2024-02-18T07:03:45.537448Z","shell.execute_reply.started":"2024-02-18T07:03:36.115202Z","shell.execute_reply":"2024-02-18T07:03:45.536182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'eeg_mean_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_min_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_max_f{x}_10s' for x in range(512)]\nFEATURES += [f'eeg_std_f{x}_10s' for x in range(512)]\n\n\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n    \n    # RESHAPE EEG SPECTROGRAMS 128x256x4 => 512x256\n    eeg_spec = np.zeros((512,256),dtype='float32')\n    xx = all_eegs2[row.eeg_id]\n    for j in range(4): eeg_spec[128*j:128*(j+1),] = xx[:,:,j]\n\n    # 10 SECOND WINDOW FROM EEG SPECTROGRAMS \n    x = np.nanmean(eeg_spec.T[100:-100,:],axis=0)\n    data[k,1600:2112] = x\n    x = np.nanmin(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2112:2624] = x\n    x = np.nanmax(eeg_spec.T[100:-100,:],axis=0)\n    data[k,2624:3136] = x\n    x = np.nanstd(eeg_spec.T[100:-100,:],axis=0)\n    data[k,3136:3648] = x\n\ntest[FEATURES] = data\nprint('New test shape',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:45.539498Z","iopub.execute_input":"2024-02-18T07:03:45.540459Z","iopub.status.idle":"2024-02-18T07:03:49.310752Z","shell.execute_reply.started":"2024-02-18T07:03:45.540409Z","shell.execute_reply":"2024-02-18T07:03:49.309763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nn_samples = test.shape[0]  # 假设test是pandas DataFrame\nn_classes = 6  # 假设有5个类加上一个\"其他\"类\npreds = np.zeros((n_samples, n_classes))  # 初始化为0\n\nfor fold in range(5):\n    print(fold, ', ', end='')\n    model1 = CatBoostClassifier(task_type='GPU')\n#     model1.load_model(f'/kaggle/input/catboost-binary-classification-other-vs-rest/CAT_v{VER}_f{fold}.cat')\n    model1.load_model(f'CAT_v{VER}_f{fold}.cat')\n\n    \n    model2 = CatBoostClassifier(task_type='GPU')\n#     model2.load_model(f'/kaggle/input/catboost-binary-classification-other-vs-rest/CAT_multi_v{VER}_f{fold}.cat')\n    model2.load_model(f'CAT_multi_v{VER}_f{fold}.cat')\n\n    \n    test_pool = Pool(data=test[FEATURES])\n    \n    prob_class_other = model1.predict_proba(test_pool)[:, 1]\n    prob_not_class_other = 1 - prob_class_other\n    \n    probs_1_to_5 = model2.predict_proba(test_pool)\n    \n    # 更新preds数组\n    for j in range(1, 6):\n        preds[:, j-1] += probs_1_to_5[:, j-1] * prob_not_class_other  # 更新1到5类的概率\n    \n    preds[:, 5] += prob_class_other  # 更新\"其他\"类的概率\n\n# 在循环结束后计算平均值\npreds /= 5\n\nprint('\\nTest preds shape:', preds.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:49.312145Z","iopub.execute_input":"2024-02-18T07:03:49.312591Z","iopub.status.idle":"2024-02-18T07:03:51.523827Z","shell.execute_reply.started":"2024-02-18T07:03:49.312553Z","shell.execute_reply":"2024-02-18T07:03:51.522751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = list(df.columns[-6:])\nsub[TARGETS] = preds\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:51.525104Z","iopub.execute_input":"2024-02-18T07:03:51.525445Z","iopub.status.idle":"2024-02-18T07:03:51.724381Z","shell.execute_reply.started":"2024-02-18T07:03:51.525416Z","shell.execute_reply":"2024-02-18T07:03:51.723225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[TARGETS].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T07:03:51.726049Z","iopub.execute_input":"2024-02-18T07:03:51.726760Z","iopub.status.idle":"2024-02-18T07:03:51.738700Z","shell.execute_reply.started":"2024-02-18T07:03:51.726721Z","shell.execute_reply":"2024-02-18T07:03:51.737527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}