{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on [CatBoost Starter - [LB 0.60]](https://www.kaggle.com/code/cdeotte/catboost-starter-lb-0-60) by [Chris Deotte](https://www.kaggle.com/cdeotte)\n\nI'm going to use run_ica function before denoisng as a first step of preprocessing. For that i transform parquet of eegs to raw for using mne library methods. But i don't understand clearly fundamental sense ICA components and how it show hidden info of data. Also i should to think about merge ICA features with something because LB score was awful (0.91LB) ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os, gc\nimport matplotlib.pyplot as plt\nimport mne\nfrom mne.preprocessing import ICA\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:00.114990Z","iopub.execute_input":"2024-03-24T12:32:00.115821Z","iopub.status.idle":"2024-03-24T12:32:03.008178Z","shell.execute_reply.started":"2024-03-24T12:32:00.115790Z","shell.execute_reply":"2024-03-24T12:32:03.007164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.009934Z","iopub.execute_input":"2024-03-24T12:32:03.010355Z","iopub.status.idle":"2024-03-24T12:32:03.310761Z","shell.execute_reply.started":"2024-03-24T12:32:03.010321Z","shell.execute_reply":"2024-03-24T12:32:03.309828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.312046Z","iopub.execute_input":"2024-03-24T12:32:03.312318Z","iopub.status.idle":"2024-03-24T12:32:03.398495Z","shell.execute_reply.started":"2024-03-24T12:32:03.312294Z","shell.execute_reply":"2024-03-24T12:32:03.397637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ch_names = ['Fp1', 'F3', 'C3', 'P3', 'F7',\n                'T3', 'T5', 'O1', 'Fz', 'Cz',\n                'Pz', 'Fp2', 'F4', 'C4', 'P4',\n                'F8', 'T4', 'T6', 'O2', 'EKG']\nsampling_freq = 200\nn_components = 15","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.401265Z","iopub.execute_input":"2024-03-24T12:32:03.401588Z","iopub.status.idle":"2024-03-24T12:32:03.406228Z","shell.execute_reply.started":"2024-03-24T12:32:03.401561Z","shell.execute_reply":"2024-03-24T12:32:03.405270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_eeg(parquet_path):\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg) - 10_000) // 2\n    eeg = eeg.iloc[middle:middle + 10_000]\n    return eeg","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.407307Z","iopub.execute_input":"2024-03-24T12:32:03.407607Z","iopub.status.idle":"2024-03-24T12:32:03.416843Z","shell.execute_reply.started":"2024-03-24T12:32:03.407582Z","shell.execute_reply":"2024-03-24T12:32:03.416080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_info():\n    sampling_freq = 200  # in Hertz\n    ch_names = ['Fp1', 'F3', 'C3', 'P3', 'F7',\n                'T3', 'T5', 'O1', 'Fz', 'Cz',\n                'Pz', 'Fp2', 'F4', 'C4', 'P4',\n                'F8', 'T4', 'T6', 'O2', 'EKG']\n    ch_types = ['eeg'] * (len(ch_names) - 1) + [\"ecg\"]  # or n_channels\n    info = mne.create_info(ch_names, ch_types=ch_types, sfreq=sampling_freq)\n    info.set_montage(\"standard_1020\")  # mne.channels.get_builtin_montages()\n    info[\"description\"] = \"HMS Harmful Activity Classification\"\n    return info","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.417860Z","iopub.execute_input":"2024-03-24T12:32:03.418120Z","iopub.status.idle":"2024-03-24T12:32:03.426771Z","shell.execute_reply.started":"2024-03-24T12:32:03.418097Z","shell.execute_reply":"2024-03-24T12:32:03.425932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_inf(eeg):\n    eeg.replace([np.inf, -np.inf], np.nan, inplace=True)\n    eeg.fillna(eeg.mean(), inplace=True)\n    return eeg","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.427790Z","iopub.execute_input":"2024-03-24T12:32:03.428112Z","iopub.status.idle":"2024-03-24T12:32:03.441061Z","shell.execute_reply.started":"2024-03-24T12:32:03.428083Z","shell.execute_reply":"2024-03-24T12:32:03.440223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_nan (eeg):\n    eeg.fillna(eeg.mean(), inplace=True)\n    return eeg","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.442066Z","iopub.execute_input":"2024-03-24T12:32:03.442633Z","iopub.status.idle":"2024-03-24T12:32:03.451687Z","shell.execute_reply.started":"2024-03-24T12:32:03.442601Z","shell.execute_reply":"2024-03-24T12:32:03.450843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_inf_nan(eeg):\n    inf_count = eeg.isin([np.inf, -np.inf]).sum()\n    nan_count = eeg.isna().sum()\n    # if inf_count.any() > 0:\n    #     print(\"*\"*15,\"Количество значений inf:\", inf_count.sum())\n    # if nan_count.any() > 0:\n    #     print(\"*\"*15,\"Количество значений NaN:\", nan_count.sum())\n    return inf_count.sum(), nan_count.sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.452776Z","iopub.execute_input":"2024-03-24T12:32:03.453073Z","iopub.status.idle":"2024-03-24T12:32:03.462081Z","shell.execute_reply.started":"2024-03-24T12:32:03.453041Z","shell.execute_reply":"2024-03-24T12:32:03.461300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_raw(eeg, info):\n    data = eeg.values.T\n    raw = mne.io.RawArray(data, info, verbose=False).crop(0, 49.995).pick('eeg')\n    raw.filter(1, 40, fir_design='firwin', verbose=50)\n    return raw","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.465239Z","iopub.execute_input":"2024-03-24T12:32:03.465570Z","iopub.status.idle":"2024-03-24T12:32:03.472390Z","shell.execute_reply.started":"2024-03-24T12:32:03.465546Z","shell.execute_reply":"2024-03-24T12:32:03.471576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_ica(raw):\n    mne.set_log_level(verbose=50, return_old_level=False, add_frames=None)\n    ica = ICA(\n        n_components=n_components,\n        noise_cov=None,\n        random_state=None,\n        method=\"fastica\",\n        fit_params=dict(tol=0.004),\n        max_iter=\"auto\",\n        allow_ref_meg=False,\n        verbose=50,\n    )\n    ica.fit(raw)\n    components = ica.get_components()\n    return components","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.473346Z","iopub.execute_input":"2024-03-24T12:32:03.473623Z","iopub.status.idle":"2024-03-24T12:32:03.483145Z","shell.execute_reply.started":"2024-03-24T12:32:03.473600Z","shell.execute_reply":"2024-03-24T12:32:03.482276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_features(eeg_id, components, df):\n    components_1d = components.flatten()\n    df.loc[len(df), 'eeg_id'] = eeg_id\n    df.iloc[len(df)-1, 1:] = components_1d","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.484004Z","iopub.execute_input":"2024-03-24T12:32:03.484271Z","iopub.status.idle":"2024-03-24T12:32:03.493577Z","shell.execute_reply.started":"2024-03-24T12:32:03.484248Z","shell.execute_reply":"2024-03-24T12:32:03.492814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Создаем DataFrame\ndf_features = pd.DataFrame()\nlist_of_features = []\nfor i in range(len(ch_names)-1):\n    for k in range(n_components):\n        list_of_features.append(f'{ch_names[i]}_component_№{k+1}')\ndf_features = pd.DataFrame(columns=['eeg_id'] + list_of_features)","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.494622Z","iopub.execute_input":"2024-03-24T12:32:03.494901Z","iopub.status.idle":"2024-03-24T12:32:03.519808Z","shell.execute_reply.started":"2024-03-24T12:32:03.494867Z","shell.execute_reply":"2024-03-24T12:32:03.519004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '../input/hms-harmful-brain-activity-classification/train_eegs/'\n#train = pd.read_csv('../input/hms-harmful-brain-activity-classification/train.csv')\nEEG_IDS = train.eeg_id.unique()\nall_components = []\ninfo = create_info()\n\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='')\n    # CREATE 50 seconds EEG FROM PARQUET\n    eeg = get_eeg(f'{PATH}{eeg_id}.parquet')\n\n    count_inf_nan(eeg)\n    \n    fill_inf (eeg)\n\n    fill_nan (eeg)\n    \n    # CREATE RAW FOR EEG \n    raw = create_raw(eeg, info)\n    \n    # CREATE ICA\n    components = run_ica(raw)\n    # all_components.append(components)\n    \n    # CREATE DataFrame with FEATURES\n    create_features(eeg_id, components, df_features)\n\ntrain.merge(df_features, how='inner', on='eeg_id')\nprint(df_features.head())\nprint(train.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-24T12:32:03.520860Z","iopub.execute_input":"2024-03-24T12:32:03.521120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(df_features, how='inner', on='eeg_id')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = df_features.columns.values[1:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import catboost as cat\nfrom catboost import CatBoostClassifier, Pool\nprint('CatBoost version',cat.__version__)\nVER = 3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\n\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train, train.target, train.patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = CatBoostClassifier(task_type='GPU',\n                               loss_function='MultiClass')\n    \n    train_pool = Pool(\n        data = train.loc[train_index,FEATURES],\n        label = train.loc[train_index,'target'].map(TARS),\n    )\n    \n    valid_pool = Pool(\n        data = train.loc[valid_index,FEATURES],\n        label = train.loc[valid_index,'target'].map(TARS),\n    )\n    \n    model.fit(train_pool,\n             verbose=100,\n             eval_set=valid_pool,\n             )\n    model.save_model(f'CAT_v{VER}_f{i}.cat')\n    \n    oof = model.predict_proba(valid_pool)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del train_pool, valid_pool, oof #model\n    gc.collect()\n    \n    #break\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOP = 25\n\nfeature_importance = model.feature_importances_\nsorted_idx = np.argsort(feature_importance)\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], np.array(FEATURES)[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH2 = '../input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 0\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\n# Создаем DataFrame\ndf_features2 = pd.DataFrame()\nlist_of_features = []\nfor i in range(len(ch_names)-1):\n    for k in range(n_components):\n        list_of_features.append(f'{ch_names[i]}_component_№{k+1}')\ndf_features2 = pd.DataFrame(columns=['eeg_id'] + list_of_features)\n\nfor i,eeg_id in enumerate(EEG_IDS2):\n    \n    eeg = get_eeg(f'{PATH2}{eeg_id}.parquet')\n\n    count_inf_nan(eeg)\n    \n    fill_inf (eeg)\n\n    fill_nan (eeg)\n    \n    # CREATE RAW FOR EEG \n    raw = create_raw(eeg, info)\n    \n    # CREATE ICA\n    components = run_ica(raw)\n    \n    # CREATE DataFrame with FEATURES\n    create_features(eeg_id, components, df_features2)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.merge(df_features2, how='inner', on='eeg_id')\nprint(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER CATBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i,', ',end='')\n    model = CatBoostClassifier(task_type='GPU')\n    model.load_model(f'CAT_v{VER}_f{i}.cat')\n    \n    test_pool = Pool(\n        data = test[FEATURES]\n    )\n    \n    pred = model.predict_proba(test_pool)\n    preds.append(pred)\npred = np.mean(preds,axis=0)\nprint()\nprint('Test preds shape',pred.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}