{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7387907,"sourceType":"datasetVersion","datasetId":4294350},{"sourceId":158715339,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"***Notebook with EEG describe features and spectrogram describe features,\n    feel free to fork this and do leave an upvote if you find this helpful :')***","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-13T05:57:49.658211Z","iopub.execute_input":"2024-01-13T05:57:49.659065Z","iopub.status.idle":"2024-01-13T05:58:05.899294Z","shell.execute_reply.started":"2024-01-13T05:57:49.659028Z","shell.execute_reply":"2024-01-13T05:58:05.898516Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport copy\nfrom sklearn.model_selection import train_test_split\n\nimport numpy as np\nimport keras\nimport os\n\nimport lightgbm as lgb\nfrom sklearn.metrics import accuracy_score\nimport xgboost as xgb","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:05.900892Z","iopub.execute_input":"2024-01-13T05:58:05.901179Z","iopub.status.idle":"2024-01-13T05:58:06.152994Z","shell.execute_reply.started":"2024-01-13T05:58:05.901155Z","shell.execute_reply":"2024-01-13T05:58:06.152018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making x_train","metadata":{}},{"cell_type":"code","source":"# %%time\n# import os\n# import concurrent.futures\n# import pandas as pd\n\n# folder_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms'\n\n# def process_file(filename):\n#     file_path = os.path.join(folder_path, filename)\n#     df = pd.read_parquet(file_path)\n#     df = df.describe().drop(df.describe().index[0])\n#     return filename.split('.')[0], df.to_numpy().flatten()\n\n# train_spectrograms = {}\n# file_list = os.listdir(folder_path)\n\n# with concurrent.futures.ProcessPoolExecutor() as executor:\n#     futures = {executor.submit(process_file, filename): filename for filename in file_list}\n#     for future in concurrent.futures.as_completed(futures):\n#         filename = futures[future]\n#         try:\n#             result = future.result()\n#             train_spectrograms[result[0]] = result[1]\n#         except Exception as e:\n#             print(f\"Error processing file {filename}: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-12T17:51:59.826723Z","iopub.execute_input":"2024-01-12T17:51:59.827009Z","iopub.status.idle":"2024-01-12T17:51:59.831452Z","shell.execute_reply.started":"2024-01-12T17:51:59.826985Z","shell.execute_reply":"2024-01-12T17:51:59.830611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import dump, load\n# dump(train_eegs,'train_eegs.joblib')\ntrain_spectrograms = load('/kaggle/input/spects/train_spectrograms.joblib')","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:06.154174Z","iopub.execute_input":"2024-01-13T05:58:06.154462Z","iopub.status.idle":"2024-01-13T05:58:09.196585Z","shell.execute_reply.started":"2024-01-13T05:58:06.154437Z","shell.execute_reply":"2024-01-13T05:58:09.19563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = train[['eeg_id','spectrogram_id','seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote','expert_consensus']]\nx_train = x_train.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.197789Z","iopub.execute_input":"2024-01-13T05:58:09.198067Z","iopub.status.idle":"2024-01-13T05:58:09.255108Z","shell.execute_reply.started":"2024-01-13T05:58:09.198042Z","shell.execute_reply":"2024-01-13T05:58:09.254175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dict of eeg_id and set of its unique spectrograms\neeg_set = {}\nfor i in x_train[['eeg_id','spectrogram_id']].values:\n    if i[0] in eeg_set:\n        if i[1] not in eeg_set[i[0]]:\n            eeg_set[i[0]] = eeg_set.get(i[0]).add(i[1])\n    else:\n        temp = set({})\n        temp.add(i[1])\n        eeg_set[i[0]] = temp","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.257947Z","iopub.execute_input":"2024-01-13T05:58:09.258331Z","iopub.status.idle":"2024-01-13T05:58:09.30229Z","shell.execute_reply.started":"2024-01-13T05:58:09.258293Z","shell.execute_reply":"2024-01-13T05:58:09.301361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dict of eeg_id and its merged spectrogram_\neeg_spect = {}\nfor i in list(eeg_set.keys()):\n    #j is spects for a 'i' eeg_id key\n    for j in list(eeg_set[i]):\n        if i in eeg_spect:\n            eeg_spect[i] = eeg_spect[i].append(train_spectrograms[str(j)])\n        else:\n            temp = []\n            temp.append(train_spectrograms[str(j)])\n            eeg_spect[i] = temp\n            \nfor i in list(eeg_spect.keys()):\n    eeg_spect[i] = np.asarray(eeg_spect[i])","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.303612Z","iopub.execute_input":"2024-01-13T05:58:09.303976Z","iopub.status.idle":"2024-01-13T05:58:09.692118Z","shell.execute_reply.started":"2024-01-13T05:58:09.303944Z","shell.execute_reply":"2024-01-13T05:58:09.691082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = []\nfor i in x_train.values:\n    key = i[0]\n    val.append(eeg_spect[key])\nval = np.asarray(val)\nval = val.reshape(-1,2807)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.693222Z","iopub.execute_input":"2024-01-13T05:58:09.693524Z","iopub.status.idle":"2024-01-13T05:58:09.871971Z","shell.execute_reply.started":"2024-01-13T05:58:09.693499Z","shell.execute_reply":"2024-01-13T05:58:09.870929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = pd.DataFrame(val, columns= range(len(val[0])))","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.873307Z","iopub.execute_input":"2024-01-13T05:58:09.874426Z","iopub.status.idle":"2024-01-13T05:58:09.879534Z","shell.execute_reply.started":"2024-01-13T05:58:09.874389Z","shell.execute_reply":"2024-01-13T05:58:09.87864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train.reset_index()\nx_train = pd.concat([x_train, n], axis=1)\nx_train = x_train.drop(columns = ['index'])","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:58:09.880898Z","iopub.execute_input":"2024-01-13T05:58:09.881213Z","iopub.status.idle":"2024-01-13T05:58:10.32899Z","shell.execute_reply.started":"2024-01-13T05:58:09.881185Z","shell.execute_reply":"2024-01-13T05:58:10.327786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_eegs = load('/kaggle/input/train-eegs/train_eegs.joblib')\n\nval = []\nfor i in x_train.values:\n    key = str(i[0])\n    val.append(train_eegs[key])\nval = np.asarray(val)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T05:59:36.868904Z","iopub.execute_input":"2024-01-13T05:59:36.869509Z","iopub.status.idle":"2024-01-13T05:59:42.013583Z","shell.execute_reply.started":"2024-01-13T05:59:36.869477Z","shell.execute_reply":"2024-01-13T05:59:42.012742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = pd.DataFrame(val, columns= range(3000,3000+len(val[0])) )\n\nx_train = x_train.reset_index()\nx_train = pd.concat([x_train, n], axis=1)\nx_train = x_train.drop(columns = ['index'])","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:00:43.468264Z","iopub.execute_input":"2024-01-13T06:00:43.46868Z","iopub.status.idle":"2024-01-13T06:00:43.938695Z","shell.execute_reply.started":"2024-01-13T06:00:43.468649Z","shell.execute_reply":"2024-01-13T06:00:43.937878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making x_test","metadata":{}},{"cell_type":"code","source":"x_test = test[['eeg_id']]\nx_sub = sub[['eeg_id']]","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:02.268418Z","iopub.execute_input":"2024-01-13T06:01:02.269262Z","iopub.status.idle":"2024-01-13T06:01:02.275171Z","shell.execute_reply.started":"2024-01-13T06:01:02.26923Z","shell.execute_reply":"2024-01-13T06:01:02.274021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport os\nimport concurrent.futures\nimport pandas as pd\n\nfolder_path = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms'\n\ndef process_file(filename):\n    file_path = os.path.join(folder_path, filename)\n    df = pd.read_parquet(file_path)\n    df = df.describe().drop(df.describe().index[0])\n    return filename.split('.')[0], df.to_numpy().flatten()\n\ntest_spectrograms = {}\nfile_list = os.listdir(folder_path)\n\nwith concurrent.futures.ProcessPoolExecutor() as executor:\n    futures = {executor.submit(process_file, filename): filename for filename in file_list}\n    for future in concurrent.futures.as_completed(futures):\n        filename = futures[future]\n        try:\n            result = future.result()\n            test_spectrograms[result[0]] = result[1]\n        except Exception as e:\n            print(f\"Error processing file {filename}: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:06.948483Z","iopub.execute_input":"2024-01-13T06:01:06.948855Z","iopub.status.idle":"2024-01-13T06:01:08.726079Z","shell.execute_reply.started":"2024-01-13T06:01:06.948827Z","shell.execute_reply":"2024-01-13T06:01:08.724852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dict of eeg_id and set of its unique containing spects\neeg_set = {}\nfor i in test[['eeg_id','spectrogram_id']].values:\n    if i[0] in eeg_set:\n        if i[1] not in eeg_set[i[0]]:\n            eeg_set[i[0]] = eeg_set.get(i[0]).add(i[1])\n    else:\n        temp = set({})\n        temp.add(i[1])\n        eeg_set[i[0]] = temp","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:08.747926Z","iopub.execute_input":"2024-01-13T06:01:08.748304Z","iopub.status.idle":"2024-01-13T06:01:08.756903Z","shell.execute_reply.started":"2024-01-13T06:01:08.74827Z","shell.execute_reply":"2024-01-13T06:01:08.75579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dict of eeg_id and its merged spectrogram_ids\neeg_spect = {}\nfor i in list(eeg_set.keys()):\n    #j is spectrogram_id in 'i' - eeg_id key\n    for j in list(eeg_set[i]):\n        if i in eeg_spect:\n            eeg_spect[i] = eeg_spect[i].append(test_spectrograms[str(j)])\n        else:\n            temp = []\n            temp.append(test_spectrograms[str(j)])\n            eeg_spect[i] = temp\n            \nfor i in list(eeg_spect.keys()):\n    eeg_spect[i] = np.asarray(eeg_spect[i])\n    if eeg_spect[i].shape[0] >1:\n        eeg_spect[i] = np.mean(eeg_spect[i], axis=0)\n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:12.168501Z","iopub.execute_input":"2024-01-13T06:01:12.168851Z","iopub.status.idle":"2024-01-13T06:01:12.175946Z","shell.execute_reply.started":"2024-01-13T06:01:12.168826Z","shell.execute_reply":"2024-01-13T06:01:12.174917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = []\nfor i in x_sub.values:\n    key = i[0]\n    val.append(eeg_spect[key])\nval = np.asarray(val)\nval = val.reshape(-1,2807)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:22.368716Z","iopub.execute_input":"2024-01-13T06:01:22.369466Z","iopub.status.idle":"2024-01-13T06:01:22.374425Z","shell.execute_reply.started":"2024-01-13T06:01:22.369438Z","shell.execute_reply":"2024-01-13T06:01:22.373431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = pd.DataFrame(val, columns= range(len(val[0])) )","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:25.488205Z","iopub.execute_input":"2024-01-13T06:01:25.488576Z","iopub.status.idle":"2024-01-13T06:01:25.493357Z","shell.execute_reply.started":"2024-01-13T06:01:25.488547Z","shell.execute_reply":"2024-01-13T06:01:25.492502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_sub = x_sub.reset_index()\nx_sub = pd.concat([x_sub, n], axis=1)\nx_sub = x_sub.drop(columns= ['index'])","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:01:26.967901Z","iopub.execute_input":"2024-01-13T06:01:26.968294Z","iopub.status.idle":"2024-01-13T06:01:26.97725Z","shell.execute_reply.started":"2024-01-13T06:01:26.968264Z","shell.execute_reply":"2024-01-13T06:01:26.976159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport os\nimport concurrent.futures\nimport pandas as pd\n\nfolder_path = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs'\n\ndef process_file(filename):\n    file_path = os.path.join(folder_path, filename)\n    df = pd.read_parquet(file_path)\n    df = df.describe().drop(df.describe().index[0])\n    return filename.split('.')[0], df.to_numpy().flatten()\n\ntest_eegs = {}\nfile_list = os.listdir(folder_path)\n\nwith concurrent.futures.ProcessPoolExecutor() as executor:\n    futures = {executor.submit(process_file, filename): filename for filename in file_list}\n    for future in concurrent.futures.as_completed(futures):\n        filename = futures[future]\n        try:\n            result = future.result()\n            test_eegs[result[0]] = result[1]\n        except Exception as e:\n            print(f\"Error processing file {filename}: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:02:26.568657Z","iopub.execute_input":"2024-01-13T06:02:26.569298Z","iopub.status.idle":"2024-01-13T06:02:26.989638Z","shell.execute_reply.started":"2024-01-13T06:02:26.56927Z","shell.execute_reply":"2024-01-13T06:02:26.988319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = []\nfor i in x_sub.values:\n    key = str(int(i[0]))\n    val.append(test_eegs[key])\nval = np.asarray(val)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:03:34.908812Z","iopub.execute_input":"2024-01-13T06:03:34.909691Z","iopub.status.idle":"2024-01-13T06:03:34.914858Z","shell.execute_reply.started":"2024-01-13T06:03:34.909652Z","shell.execute_reply":"2024-01-13T06:03:34.913839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = pd.DataFrame(val, columns= range(3000,3000+len(val[0])) )\n\nx_sub = x_sub.reset_index()\nx_sub = pd.concat([x_sub, n], axis=1)\nx_sub = x_sub.drop(columns= ['index'])","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:04:04.468813Z","iopub.execute_input":"2024-01-13T06:04:04.469248Z","iopub.status.idle":"2024-01-13T06:04:04.481272Z","shell.execute_reply.started":"2024-01-13T06:04:04.469218Z","shell.execute_reply":"2024-01-13T06:04:04.480137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# data labeling and preprocessing","metadata":{}},{"cell_type":"code","source":"x_train['expert_consensus'] = x_train['expert_consensus'].replace({'Seizure': 0, 'LPD': 1, 'GPD': 2,'LRDA':3, 'GRDA':4, 'Other':5 })","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:04:47.400065Z","iopub.execute_input":"2024-01-13T06:04:47.400478Z","iopub.status.idle":"2024-01-13T06:04:47.425854Z","shell.execute_reply.started":"2024-01-13T06:04:47.400447Z","shell.execute_reply":"2024-01-13T06:04:47.425054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_train = x_train.drop(columns = ['eeg_id','spectrogram_id','seizure_vote','lpd_vote',\t'gpd_vote',\t'lrda_vote','grda_vote','other_vote','expert_consensus']).copy()\nlgb_test = x_sub.drop(columns = ['eeg_id']).copy()","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:04:50.587897Z","iopub.execute_input":"2024-01-13T06:04:50.588772Z","iopub.status.idle":"2024-01-13T06:04:51.195189Z","shell.execute_reply.started":"2024-01-13T06:04:50.588736Z","shell.execute_reply":"2024-01-13T06:04:51.194393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LGBM Modelling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\nlgb_train = pd.DataFrame(scaler.fit_transform(lgb_train), columns=lgb_train.columns)\nlgb_test = pd.DataFrame(scaler.transform(lgb_test), columns=lgb_test.columns)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:04:53.828196Z","iopub.execute_input":"2024-01-13T06:04:53.830908Z","iopub.status.idle":"2024-01-13T06:04:54.47407Z","shell.execute_reply.started":"2024-01-13T06:04:53.830863Z","shell.execute_reply":"2024-01-13T06:04:54.473105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, log_loss\n\n\nparams = {\n    'objective': 'multiclass',\n    'num_class': 6,\n    'boosting_type': 'gbdt',\n    'metric': 'multi_logloss',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5,\n    'verbose': 0,\n    'max_depth': 17,\n    'min_data_in_leaf': 20,\n    'device': 'gpu',\n}\n\n\nlgb_model = lgb.LGBMClassifier(**params)\nlgb_model.fit(lgb_train, x_train.expert_consensus , verbose=0)\n\ny_pred = lgb_model.predict(lgb_test)\ny_pred_proba = lgb_model.predict_proba(lgb_test)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-01-13T06:08:56.388646Z","iopub.execute_input":"2024-01-13T06:08:56.389029Z","iopub.status.idle":"2024-01-13T06:10:35.977382Z","shell.execute_reply.started":"2024-01-13T06:08:56.389001Z","shell.execute_reply":"2024-01-13T06:10:35.976452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['seizure_vote'] = y_pred_proba[:,0]\nsub['lpd_vote'] = y_pred_proba[:,1]\nsub['gpd_vote'] = y_pred_proba[:,2]\nsub['lrda_vote'] = y_pred_proba[:,3]\nsub['grda_vote'] = y_pred_proba[:,4]\nsub['other_vote'] = y_pred_proba[:,5]","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:11:17.828381Z","iopub.execute_input":"2024-01-13T06:11:17.82907Z","iopub.status.idle":"2024-01-13T06:11:17.836188Z","shell.execute_reply.started":"2024-01-13T06:11:17.829039Z","shell.execute_reply":"2024-01-13T06:11:17.835304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T06:11:19.707685Z","iopub.execute_input":"2024-01-13T06:11:19.708532Z","iopub.status.idle":"2024-01-13T06:11:19.715524Z","shell.execute_reply.started":"2024-01-13T06:11:19.708497Z","shell.execute_reply":"2024-01-13T06:11:19.714635Z"},"trusted":true},"execution_count":null,"outputs":[]}]}