{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install mne_icalabel","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:52:49.273363Z","iopub.execute_input":"2025-03-02T12:52:49.273666Z","iopub.status.idle":"2025-03-02T12:52:56.056342Z","shell.execute_reply.started":"2025-03-02T12:52:49.273633Z","shell.execute_reply":"2025-03-02T12:52:56.055442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib\nmatplotlib.use('agg')  \nimport matplotlib.pyplot as plt\nimport mne\nfrom mne_icalabel import label_components\nfrom scipy.stats import pearsonr\nimport pywt\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:53:01.061773Z","iopub.execute_input":"2025-03-02T12:53:01.062076Z","iopub.status.idle":"2025-03-02T12:53:04.781082Z","shell.execute_reply.started":"2025-03-02T12:53:01.062049Z","shell.execute_reply":"2025-03-02T12:53:04.780071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:53:09.139089Z","iopub.execute_input":"2025-03-02T12:53:09.139612Z","iopub.status.idle":"2025-03-02T12:53:09.407466Z","shell.execute_reply.started":"2025-03-02T12:53:09.139579Z","shell.execute_reply":"2025-03-02T12:53:09.406723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\noutput_dir1 = '/kaggle/working/preprocessed'\noutput_dir2 = '/kaggle/working/coefficients'\nos.makedirs(output_dir1, exist_ok = True)\nos.makedirs(output_dir2, exist_ok = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:56:39.014225Z","iopub.execute_input":"2025-03-02T12:56:39.014596Z","iopub.status.idle":"2025-03-02T12:56:39.019155Z","shell.execute_reply.started":"2025-03-02T12:56:39.01457Z","shell.execute_reply":"2025-03-02T12:56:39.018352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ch_names = ['Fp1','F3','C3','P3','F7','T3','T5','O1','Fz','Cz','Pz','Fp2','F4','C4','P4','F8','T4','T6','O2']\nch_types = ['eeg'] * len(ch_names)\nsr = 200\n\ndef signal_processing(a,b):\n    eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/\"+str(a)+\".parquet\")\n    eeg = eeg[int(b)*200:int(50+b)*200:5] #[start, stop, step]\n\n    eeg.fillna(eeg.mean(), inplace=True)\n    \n    # eeg.drop(columns=['EKG'], inplace=True)\n    ekg_data = eeg['EKG'] \n    eeg_array = eeg.iloc[:,:-1].to_numpy().T \n    \n    info = mne.create_info(ch_names=ch_names, sfreq=sr, ch_types=ch_types)\n    raw = mne.io.RawArray(eeg_array, info)\n    montage = mne.channels.make_standard_montage('standard_1020')\n    raw.set_montage(montage)\n\n    raw.notch_filter(60)\n\n    raw.filter(l_freq=1, h_freq=99.99, picks='data', method='iir', iir_params=None)\n\n    return raw, ekg_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:56:57.886455Z","iopub.execute_input":"2025-03-02T12:56:57.886775Z","iopub.status.idle":"2025-03-02T12:56:57.893456Z","shell.execute_reply.started":"2025-03-02T12:56:57.886751Z","shell.execute_reply":"2025-03-02T12:56:57.892698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_brain_labels = ['muscle artifact', 'eye blink', 'heart beat', 'line noise', 'channel noise', 'other']\n\ndef apply_ica(raw, ekg_data):\n    ica = mne.preprocessing.ICA(n_components=18, random_state=0, method='infomax', fit_params=dict(extended=True))\n    ica.fit(raw.copy())\n\n    ica_sources = ica.get_sources(raw).get_data()\n\n    CORR_THRESHOLD = 0.9  \n    ekg_indices = []\n    for i in range(ica_sources.shape[0]): \n        corr, _ = pearsonr(ica_sources[i, :], ekg_data)\n        if abs(corr) > CORR_THRESHOLD:\n            ekg_indices.append(i)\n\n    labels = label_components(raw, ica, method='iclabel')\n\n    non_brain_indices = [idx for idx, label in enumerate(labels[\"labels\"]) if label in non_brain_labels]\n    total_indices = set(range(19))\n    total_exclude = list(set(non_brain_indices + ekg_indices))\n    brain_indices = list(total_indices - set(total_exclude))\n    \n    raw = ica.apply(raw, include=brain_indices, exclude=total_exclude)\n\n    return pd.DataFrame(raw.get_data().T, columns = ch_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:03.343556Z","iopub.execute_input":"2025-03-02T12:57:03.343843Z","iopub.status.idle":"2025-03-02T12:57:03.350203Z","shell.execute_reply.started":"2025-03-02T12:57:03.343821Z","shell.execute_reply":"2025-03-02T12:57:03.349164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def maddest(d):\n    return np.median(np.abs(d))\n    \ndef single_channel_dwt(x, wavelet='db38', level=4):\n    cf = ['cA4', 'cD4', 'cD3', 'cD2', 'cD1'] \n    ret = {key:[] for key in x.columns} \n\n    for pos in x.columns:\n        coeff = pywt.wavedec(x[pos], wavelet, mode=\"per\", level=level)\n        \n        for j in range(0,len(cf)): \n            sigma = (1/0.6745) * maddest(coeff[j])\n            uthresh = sigma * np.sqrt(2*np.log(len(x))) \n            coeff[j] = pywt.threshold(coeff[j], value=uthresh, mode='soft')\n\n        ret[pos]=pywt.waverec(coeff, wavelet, mode='per')\n\n    return pd.DataFrame(ret)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:08.106773Z","iopub.execute_input":"2025-03-02T12:57:08.107085Z","iopub.status.idle":"2025-03-02T12:57:08.112628Z","shell.execute_reply.started":"2025-03-02T12:57:08.107059Z","shell.execute_reply":"2025-03-02T12:57:08.111756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bi_channel_dwt(x, wavelet='db38', level=4):\n    cf = ['cA4', 'cD4', 'cD3', 'cD2', 'cD1'] \n    ret = {key:[] for key in x.columns} \n    l = {key:[] for key in cf} \n    \n    for pos in x.columns:\n        coeff = pywt.wavedec(x[pos], wavelet, mode=\"per\", level=level)\n        # l[cf[0]].append(coeff[0])\n\n        for j in range(0,len(l)): \n            sigma = (1/0.6745) * maddest(coeff[j])\n            uthresh = sigma * np.sqrt(2*np.log(len(x))) \n            coeff[j] = pywt.threshold(coeff[j], value=uthresh, mode='soft')\n            l[cf[j]].append(coeff[j])\n\n        ret[pos]=pywt.waverec(coeff, wavelet, mode='per')\n\n    avg = {}\n    for a in l.keys():\n       avg[a]  = np.mean(l[a],axis=0) \n        \n    return pd.DataFrame(ret), avg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:12.976208Z","iopub.execute_input":"2025-03-02T12:57:12.976544Z","iopub.status.idle":"2025-03-02T12:57:12.982533Z","shell.execute_reply.started":"2025-03-02T12:57:12.976516Z","shell.execute_reply":"2025-03-02T12:57:12.981745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_eeg_ids = np.sort(df[\"eeg_id\"].unique())\none_fourth = len(unique_eeg_ids) // 4\nhalf = len(unique_eeg_ids) // 2\nthree_fourth = 3 * len(unique_eeg_ids) // 4\n\none_fourth_eeg_ids = unique_eeg_ids[: one_fourth]\ntwo_fourth_eeg_ids = unique_eeg_ids[one_fourth: half]\nthree_fourth_eeg_ids = unique_eeg_ids[half: three_fourth]\nfour_fourth_eeg_ids = unique_eeg_ids[three_fourth: ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:19.469303Z","iopub.execute_input":"2025-03-02T12:57:19.469756Z","iopub.status.idle":"2025-03-02T12:57:19.48202Z","shell.execute_reply.started":"2025-03-02T12:57:19.46972Z","shell.execute_reply":"2025-03-02T12:57:19.481069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = df['expert_consensus'].unique()\nlabel_num = {}\nnum_label = {}\n\nfor i,j in enumerate(labels):\n    label_num[j] = i\n    num_label[i] = j","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:25.598745Z","iopub.execute_input":"2025-03-02T12:57:25.599027Z","iopub.status.idle":"2025-03-02T12:57:25.611484Z","shell.execute_reply.started":"2025-03-02T12:57:25.599006Z","shell.execute_reply":"2025-03-02T12:57:25.610671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"index=[[0,4],[4,5],[5,6],[6,7],[11,15],[15,16],[16,17],[17,18],[0,1],[1,2],[2,3],[3,7],[11,12],[12,13],[13,14],[14,18],[8,9],[9,10]]\na=0\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nmne.set_log_level(\"WARNING\") \n\nfor k, eeg_id in enumerate(tqdm(one_fourth_eeg_ids)):\n    eeg_data = df[df[\"eeg_id\"] == eeg_id]\n    \n    for _, row in eeg_data.iterrows():\n        a = row[\"eeg_id\"] \n        b = int(row[\"eeg_label_offset_seconds\"])  \n        c = row[\"eeg_sub_id\"] \n        lbl = label_num[row['expert_consensus']]\n        \n        raw, ekg_data = signal_processing(a, b)\n        artifact_corrected_data = apply_ica(raw, ekg_data)\n        sc = single_channel_dwt(artifact_corrected_data)\n        \n        banana_mon = pd.DataFrame()\n        for i, j in index:\n            banana_mon[sc.columns[i] + '-' + sc.columns[j]] = sc[sc.columns[i]] - sc[sc.columns[j]]\n        bi, coeff = bi_channel_dwt(banana_mon)\n\n        bi_array = bi.to_numpy().T\n        # final_array = (bi_array, lbl)\n        \n        np.savez(f\"/kaggle/working/preprocessed/{a}_{c}_signal.npy\", eeg = bi_array, label=np.array([lbl])) \n        np.save(f\"/kaggle/working/coefficients/{a}_{c}_coeff.npy\", coeff)\n\n    if (k + 1) % 100 == 0:\n        print(k)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T12:57:32.268873Z","iopub.execute_input":"2025-03-02T12:57:32.269147Z","execution_failed":"2025-03-03T00:52:46.071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink \nFileLink(r'preprocessed.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-03T03:56:27.45304Z","iopub.execute_input":"2025-03-03T03:56:27.45341Z","iopub.status.idle":"2025-03-03T03:56:27.460035Z","shell.execute_reply.started":"2025-03-03T03:56:27.45338Z","shell.execute_reply":"2025-03-03T03:56:27.459111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}