{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Same as [CatBoost Starter - [LB 0.67]](https://www.kaggle.com/code/cdeotte/catboost-starter-lb-0-67)\n\n**V3**\n- Modified targets as Other vs All Seizures\n\n**V5**\n- Modified target weights as 0.95 and 0.05","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\n\nVER = 2","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:49.073406Z","iopub.execute_input":"2024-01-20T01:25:49.074059Z","iopub.status.idle":"2024-01-20T01:25:49.413617Z","shell.execute_reply.started":"2024-01-20T01:25:49.074025Z","shell.execute_reply":"2024-01-20T01:25:49.412777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf['expert_consensus'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:49.442656Z","iopub.execute_input":"2024-01-20T01:25:49.443025Z","iopub.status.idle":"2024-01-20T01:25:50.060535Z","shell.execute_reply.started":"2024-01-20T01:25:49.443000Z","shell.execute_reply":"2024-01-20T01:25:50.059408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = list(df.columns[-6:])\nTARGETS_GROUP2 = ['other_vote']\nTARGETS_GROUP1 = list(set(list(df.columns[-6:])) - set(TARGETS_GROUP2))\ndf['all_target_group1'] = df.apply(lambda row: sum([row[x] for x in TARGETS_GROUP1]) , axis=1)\ndf['all_target_group2'] = df.apply(lambda row: sum([row[x] for x in TARGETS_GROUP2]) , axis=1)\nTARGETS = ['all_target_group1','all_target_group2']\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-20T01:25:50.078592Z","iopub.execute_input":"2024-01-20T01:25:50.078863Z","iopub.status.idle":"2024-01-20T01:25:54.596690Z","shell.execute_reply.started":"2024-01-20T01:25:50.078839Z","shell.execute_reply":"2024-01-20T01:25:54.595726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Non-Overlapping Eeg Id Train Data\nThe competition data description says that test data does not have multiple crops from the same `eeg_id`. Therefore we will train and validate using only 1 crop per `eeg_id`. There is a discussion about this [here][1].\n\n[1]: https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/discussion/467021","metadata":{}},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].apply(lambda x: 0.95 if x >= 0.5 else 0.05)\ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['org_target'] = tmp\n\nTARGETS_GROUP1_NAMES = [x.lower().replace(\"_vote\",\"\").strip() for x in TARGETS_GROUP1]\ntrain['target'] = train.apply(lambda x: \"Group1\" if  x['org_target'].lower() in  TARGETS_GROUP1_NAMES else \"Group2\", axis=1)\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\n\ntrain['all_target_group1'] = train.apply(lambda x: 0.95 if x['target'] == 'Group1' else 0.05, axis=1)\ntrain['all_target_group2'] = train.apply(lambda x: 0.95 if x['target'] == 'Group2' else 0.05, axis=1)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:54.598390Z","iopub.execute_input":"2024-01-20T01:25:54.598682Z","iopub.status.idle":"2024-01-20T01:25:55.232387Z","shell.execute_reply.started":"2024-01-20T01:25:54.598658Z","shell.execute_reply":"2024-01-20T01:25:55.231366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['org_target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:55.233839Z","iopub.execute_input":"2024-01-20T01:25:55.234537Z","iopub.status.idle":"2024-01-20T01:25:55.442870Z","shell.execute_reply.started":"2024-01-20T01:25:55.234500Z","shell.execute_reply":"2024-01-20T01:25:55.441989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:55.445364Z","iopub.execute_input":"2024-01-20T01:25:55.446028Z","iopub.status.idle":"2024-01-20T01:25:55.612224Z","shell.execute_reply.started":"2024-01-20T01:25:55.445993Z","shell.execute_reply":"2024-01-20T01:25:55.611316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nIn this section, we create features for our CatBoost model. \n\nFirst we need to read in all 11k train spectrogram files. Reading thousands of files takes 11 minutes with Pandas. Instead, we can read 1 file from my [Kaggle dataset here][1] which contains all the 11k spectrograms in less than 1 minute! To use my [Kaggle dataset][1], set variable `READ_SPEC_FILES = False`. Don't forget to upvote this helpful [dataset][1] :-)\n\nNext we need to engineer features for our CatBoost model. In this notebook, we just take the mean (over time) of each of the 400 spectrogram frequencies (using middle 10 minutes). This produces 400 features (per each unique eeg id). We can improve CV and LB score by engineering new features (and/or tuning CatBoost).\n\nUPDATE: Version 2 creates features from `means` and `mins`. And version 2 uses `10 minute windows` and `20 second windows`.\n\n[1]: https://www.kaggle.com/datasets/cdeotte/brain-spectrograms","metadata":{}},{"cell_type":"code","source":"READ_SPEC_FILES = False\nFEATURE_ENGINEER = True","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:55.613453Z","iopub.execute_input":"2024-01-20T01:25:55.614482Z","iopub.status.idle":"2024-01-20T01:25:55.618370Z","shell.execute_reply.started":"2024-01-20T01:25:55.614447Z","shell.execute_reply":"2024-01-20T01:25:55.617526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES: \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:25:55.619527Z","iopub.execute_input":"2024-01-20T01:25:55.620146Z","iopub.status.idle":"2024-01-20T01:26:48.281008Z","shell.execute_reply.started":"2024-01-20T01:25:55.620114Z","shell.execute_reply":"2024-01-20T01:26:48.279922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\n\n# FEATURE NAMES\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\nif FEATURE_ENGINEER:\n    data = np.zeros((len(train),len(FEATURES)))\n    for k in tqdm(range(len(train)),  total=len(train)):\n        #if k%100==0: print(k,', ',end='')\n        row = train.iloc[k]\n        r = int( (row['min'] + row['max'])//4 ) \n        \n        # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,:400] = x\n        x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,400:800] = x\n        \n        # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,800:1200] = x\n        x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,1200:1600] = x\n\n    train[FEATURES] = data\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\nprint()\nprint('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:26:48.282318Z","iopub.execute_input":"2024-01-20T01:26:48.282732Z","iopub.status.idle":"2024-01-20T01:27:06.161850Z","shell.execute_reply.started":"2024-01-20T01:26:48.282695Z","shell.execute_reply":"2024-01-20T01:27:06.160877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:27:06.163110Z","iopub.execute_input":"2024-01-20T01:27:06.163477Z","iopub.status.idle":"2024-01-20T01:27:06.217434Z","shell.execute_reply.started":"2024-01-20T01:27:06.163444Z","shell.execute_reply":"2024-01-20T01:27:06.216595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train CatBoost\nWe use the default settings for CatBoost which are pretty good. We can tune CatBoost manually to improve CV and LB score. Note that CatBoost will automatically use both Kaggle T4 GPUs (when we add parameter `task_type='GPU'`)  for super fast training!","metadata":{}},{"cell_type":"code","source":"import catboost as cat, gc\nfrom catboost import CatBoostClassifier, Pool\nprint('CatBoost version',cat.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:27:06.218436Z","iopub.execute_input":"2024-01-20T01:27:06.218712Z","iopub.status.idle":"2024-01-20T01:27:07.215234Z","shell.execute_reply.started":"2024-01-20T01:27:06.218689Z","shell.execute_reply":"2024-01-20T01:27:07.214308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[TARGETS]","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:27:07.218921Z","iopub.execute_input":"2024-01-20T01:27:07.219255Z","iopub.status.idle":"2024-01-20T01:27:07.233306Z","shell.execute_reply.started":"2024-01-20T01:27:07.219230Z","shell.execute_reply":"2024-01-20T01:27:07.232333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\n\nall_oof = []\nall_true = []\nTARS = {'Group1':0, 'Group2':1}\n\ngkf = GroupKFold(n_splits=5)\nall_index = []\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_v{VER}_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    all_index.extend(valid_index)\n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:27:07.234843Z","iopub.execute_input":"2024-01-20T01:27:07.235128Z","iopub.status.idle":"2024-01-20T01:28:57.272682Z","shell.execute_reply.started":"2024-01-20T01:27:07.235104Z","shell.execute_reply":"2024-01-20T01:28:57.271592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance\nBelow we display the CatBoost top 25 feature importance for the last fold we trained.","metadata":{}},{"cell_type":"code","source":"TOP = 25\n\nfeature_importance = model.feature_importances_\nsorted_idx = np.argsort(feature_importance)\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], np.array(FEATURES)[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:28:57.274061Z","iopub.execute_input":"2024-01-20T01:28:57.274359Z","iopub.status.idle":"2024-01-20T01:28:57.836646Z","shell.execute_reply.started":"2024-01-20T01:28:57.274332Z","shell.execute_reply":"2024-01-20T01:28:57.834604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_all_true_classes = np.argmax(train.loc[valid_index, TARGETS].values, axis=1)\ncheck_all_true_classes","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:28:57.839341Z","iopub.execute_input":"2024-01-20T01:28:57.839690Z","iopub.status.idle":"2024-01-20T01:28:57.849257Z","shell.execute_reply.started":"2024-01-20T01:28:57.839660Z","shell.execute_reply":"2024-01-20T01:28:57.848311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[valid_index].head(3)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:28:57.850427Z","iopub.execute_input":"2024-01-20T01:28:57.851221Z","iopub.status.idle":"2024-01-20T01:28:57.985482Z","shell.execute_reply.started":"2024-01-20T01:28:57.851187Z","shell.execute_reply":"2024-01-20T01:28:57.984609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mappings = {'Group1':0, 'Group2': 1}\nall_true_classes = [mappings[x] for x in list(train.iloc[valid_index]['target'])]\nstr(all_true_classes[0:3])","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:31:58.930181Z","iopub.execute_input":"2024-01-20T01:31:58.930530Z","iopub.status.idle":"2024-01-20T01:31:59.003792Z","shell.execute_reply.started":"2024-01-20T01:31:58.930505Z","shell.execute_reply":"2024-01-20T01:31:59.002906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, f1_score, log_loss\nfrom sklearn.metrics import confusion_matrix\n\nall_oof_classes = np.argmax(all_oof, axis=1)\nall_true_classes = np.argmax(all_true, axis=1)\naccuracy = accuracy_score(all_true_classes, all_oof_classes)\n\nf1 = f1_score(all_true_classes, all_oof_classes, average='weighted')\naccuracy, f1","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:00.261028Z","iopub.execute_input":"2024-01-20T01:32:00.261742Z","iopub.status.idle":"2024-01-20T01:32:00.281823Z","shell.execute_reply.started":"2024-01-20T01:32:00.261709Z","shell.execute_reply":"2024-01-20T01:32:00.280886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix = confusion_matrix(all_true_classes, all_oof_classes)\nTARGETS_NAMES = [','.join([x.replace(\"_vote\", \" \") for x in TARGETS_GROUP1]), ','.join([x.replace(\"_vote\", \" \") for x in TARGETS_GROUP2])]\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', \n            xticklabels=TARGETS_NAMES, yticklabels=TARGETS_NAMES)\nplt.title('Can we split well?')\nplt.xlabel('Predicted Target')\nplt.ylabel('True Target')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:01.181486Z","iopub.execute_input":"2024-01-20T01:32:01.182417Z","iopub.status.idle":"2024-01-20T01:32:01.411697Z","shell.execute_reply.started":"2024-01-20T01:32:01.182382Z","shell.execute_reply":"2024-01-20T01:32:01.410684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nCounter(all_true_classes), Counter(all_oof_classes)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:02.713348Z","iopub.execute_input":"2024-01-20T01:32:02.713733Z","iopub.status.idle":"2024-01-20T01:32:02.729928Z","shell.execute_reply.started":"2024-01-20T01:32:02.713703Z","shell.execute_reply":"2024-01-20T01:32:02.729051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"focus_ids = []\nfor i, (true_class, oof_class) in  enumerate(zip(all_true_classes, all_oof_classes)):\n    if true_class == 0 and oof_class == 1:\n        focus_ids.append(all_index[i])","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:04.034520Z","iopub.execute_input":"2024-01-20T01:32:04.034904Z","iopub.status.idle":"2024-01-20T01:32:04.050411Z","shell.execute_reply.started":"2024-01-20T01:32:04.034874Z","shell.execute_reply":"2024-01-20T01:32:04.049448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(focus_ids)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:05.324982Z","iopub.execute_input":"2024-01-20T01:32:05.325864Z","iopub.status.idle":"2024-01-20T01:32:05.331572Z","shell.execute_reply.started":"2024-01-20T01:32:05.325828Z","shell.execute_reply":"2024-01-20T01:32:05.330713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[focus_ids]['org_target'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:06.227073Z","iopub.execute_input":"2024-01-20T01:32:06.227446Z","iopub.status.idle":"2024-01-20T01:32:06.475493Z","shell.execute_reply.started":"2024-01-20T01:32:06.227417Z","shell.execute_reply":"2024-01-20T01:32:06.474601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_accuracy = 0\nfor thresold in range(0,100):\n    t = thresold/100\n    temp_all_oof = [1]*len(all_oof)\n    for i in range(len(temp_all_oof)):\n        if all_oof[i][1] > t:\n            temp_all_oof[i] = 1\n        else:\n            temp_all_oof[i] = 0\n    all_oof_classes = temp_all_oof\n    all_true_classes = np.argmax(all_true, axis=1)\n    accuracy = accuracy_score(all_true_classes, all_oof_classes)\n    if accuracy > best_accuracy:\n        best_accuracy = accuracy\n        print(\"Best found\", t, best_accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-01-20T01:32:10.850344Z","iopub.execute_input":"2024-01-20T01:32:10.851088Z","iopub.status.idle":"2024-01-20T01:32:12.710003Z","shell.execute_reply.started":"2024-01-20T01:32:10.851055Z","shell.execute_reply":"2024-01-20T01:32:12.709070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}