{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><img src=\"https://www.kaggle.com/competitions/59093/images/header\" alt=\"EEG ANALYSIS STEP\" style=\"width: 80%;\"></center>\n<br style=\"margin: 15px;\">\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-transform: none; letter-spacing: 2px; color: #F6416C;\">\n    <span style=\"text-decoration: underline;\">\n        <font color=\"#F07B3F\">H</font>MS\n        <font color=\"#F07B3F\">H</font>armful\n        <font color=\"#F07B3F\">B</font>rain\n        <font color=\"#F07B3F\">A</font>ctivity\n        <font color=\"#F07B3F\">C</font>lassification\n    </span><br><br><br style=\"margin: 15px;\">\n    <span style=\"font-size: 18px; letter-spacing: 1px;\">\n        <font color=\"#F07B3F\">E</font>XPLORATORY\n        <font color=\"#F07B3F\">D</font>ATA\n        <font color=\"#F07B3F\">A</font>NALYSIS\n        <font color=\"#F07B3F\">&nbsp;&nbsp;(EDA)</font>\n        <br style=\"margin: 18px;\">+<br style=\"margin: 18px;\">\n        <font color=\"#F07B3F\">F</font>EATURE\n        <font color=\"#F07B3F\">E</font>NGINERING\n        <br style=\"margin: 18px;\">+<br style=\"margin: 18px;\">\n        <font color=\"#F07B3F\">M</font>ODEL\n        <br style=\"margin: 15px;\">\n    </span><br style=\"margin: 15px;\">\n</h2>\n\n<p style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: #F6416C;\">CREATED BY: HAFIZ RAHMAN</p>\n\n<br>\n\n<sub><b>🎨 Mostly Font Color that i choose. (<font color=\"#FFDE7D\">#FFDE7D</font> & <font color=\"#F6416C\">#F6416C</font>) 🎨<br></b></sub>\n<br>\n\n\n\n<center><div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">⚠️ &nbsp; NOTE &nbsp; ⚠️</b><br><br><b>This notebook's primary purpose is to educate through exploration of this competition's data, the background information, and other notebooks and discussion posts that help in building a complete understanding.</b><br><br>Kaggle user's Notebooks and Discussions that I include will be listed below!\n\n<table class=\"alert alert-block alert-info\" style=\"text-align:center;\">\n<thead>\n  <tr>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Contributor&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n    <th>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;Notebook&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</th>\n  </tr>\n</thead>\n<tbody>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/lazygene\">Leslie Maltsev </a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/lazygene/visualising-pre-processed-eeg-data\">Visualising pre-processed EEG data</a></b></td>  \n  </tr>\n  <tr>\n      <td><b><a href=\"https://www.kaggle.com/mvvppp\">Vadym Malitskyi</a></b></td>\n      <td><b><a href=\"https://www.kaggle.com/code/mvvppp/hms-eda-and-domain-journey\">🌩️HMS - EDA and Domain Journey🌍</a></b></td>\n  </tr>\n  <tr>\n    <td><b><a href=\"https://www.kaggle.com/dschettler8845\">DARIEN SCHETTLER</a></b></td>\n    <td><b><a href=\"https://www.kaggle.com/code/dschettler8845/sr-rna-reactivity-learn-eda-baseline\">SR RNA Reactivity - 📚Learn – 🗺️EDA – 🤖Baseline</a></b></td>\n  </tr>\n</tbody>\n</table>\n    \n<br>\n    \n</div></center>\n\n\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #F6416C;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#introduction\" style=\"color: #F07B3F;\">1&nbsp;&nbsp;&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#competition_background_information\" style=\"color: #F07B3F;\">2&nbsp;&nbsp;&nbsp;&nbsp;HOST PROVIDED COMPETITION INFORMATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#my_understanding\" style=\"color: #F07B3F;\">3&nbsp;&nbsp;&nbsp;&nbsp;MY UNDERSTANDING OF THE COMPETITION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#domain_knowledge\" style=\"color: #F07B3F;\">4&nbsp;&nbsp;&nbsp;&nbsp;DOMAIN KNOWLEDGE</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#imports\" style=\"color: #F07B3F;\">5&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#setup\" style=\"color: #F07B3F;\">6&nbsp;&nbsp;&nbsp;&nbsp;SETUP & HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#data_exploration\" style=\"color: #F07B3F;\">7&nbsp;&nbsp;&nbsp;&nbsp;DATA EXPLORATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#tbd\" style=\"color: #F07B3F;\">8&nbsp;&nbsp;&nbsp;&nbsp;TBD</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#next_steps\" style=\"color: #F07B3F;\">9&nbsp;&nbsp;&nbsp;&nbsp;NEXT STEPS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #000;\"><a href=\"#baseline\" style=\"color: #F07B3F;\">10&nbsp;&nbsp;&nbsp;&nbsp;BASELINE</a></h3>\n\n---\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"domain knowledege\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #F07B3F;\" id=\"domain knowledge\">4&nbsp;&nbsp;DOMAIN KNOWLEDGE&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">4.1 WHAT IS EEG AND WHY SO IMPORTANT</h3>\n\nEEG is a <b><mark>noninvasive neuroimaging technique that involves the placement of electrodes on the scalp to record electrical activity of the brain</mark></b>. This enables researchers to measure and analyze the electrical signals generated by the brain. These signals offer valuable information on the operating mechanisms of the brain, covering the identification of various neurological disorders and the exploration of cognitive processes such as perception, attention, and memory. EEG has gained widespread popularity as a means of investigating electrical activity of the human brain, due to its noninvasive and safe characteristics. In addition, EEG signals have the potential to be integrated with other imaging modalities, including magnetic resonance imaging (MRI), functional near-infrared spectroscopy (fNIRS), and positron emission tomography (PET), in order to achieve a better understanding of brain function and structure.\n\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">4.2 PIPELINE OF EEG ANALYSIS</h3>\n\n<center><img src=\"https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10385593/bin/sensors-23-06434-g002.jpg\" alt=\"EEG ANALYSIS STEP\" style=\"width: 80%;\"></center>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">4.2.1 DENOISING EEG</h3>\n\n<center><img src=\"https://cdnintech.com/media/chapter/54606/1512345123/media/F2.png\" alt=\"Artifact on eeg signals\" style=\"width: 80%;\"></center>\n\nAs mentioned above regarding the acquisition of EEG signals, multiple electrodes are placed on the scalp. However, external interference can cause diverse artifacts to emerge, which can compromise the quality of the signals. Physiological artifacts, such as involuntary eye movements, blinking, heart activity, and muscle movement, are known to be present in EEG signals and can negatively affect their quality. Therefore, denoising EEG signals has become a topic of significant research interest and attention. To ensure the reliability of features extracted from EEG signals, it is essential to remove any associated artifacts. Currently, several denoising techniques have been developed such as regression method, blind source separation, wavelet transform.\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">4.2.2 Feature</h3>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">4.X ARTICLE</h3>\nhttps://ar.iub.edu.bd/bitstream/handle/123456789/560/fncom-17-1151852.pdf?sequence=1","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #F07B3F;\" id=\"imports\">5&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>\n","metadata":{}},{"cell_type":"code","source":"# NOTE : Not all packages or libraries in this code are used for this notebook; this code is from another notebook of mine\n\n%autosave 1\nclass color_class:\n    BOLD_YELLOW = '\\033[1m\\033[93m'\n    BOLD_RED = '\\033[1m\\033[91m'\n    END = '\\033[0m'\n\nprint(color_class.BOLD_YELLOW + '\\nImporting all the required libraries....\\n\\n'+ color_class.END)\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport re\nimport string\nimport glob\nimport math\nfrom IPython.display import display_html\nimport tqdm\n!pip install wandb\nimport wandb\nimport gc\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport matplotlib.patches as patches\nimport matplotlib.cm as cm\nimport seaborn as sns\nfrom plotly.offline import init_notebook_mode, iplot\nfrom plotly import tools\nimport plotly.graph_objs as go\n!pip install pywaffle\nfrom pywaffle import Waffle\nimport mne\nfrom mne.channels import make_standard_montage\nfrom mne.io import RawArray\n\n\nimport statsmodels.api as sm\nfrom scipy.stats import kurtosis, skew\n\nfrom sklearn.model_selection import (train_test_split,\n                                     cross_val_score,\n                                     StratifiedKFold,\n                                     GridSearchCV)\n\n\nfrom sklearn.preprocessing import (StandardScaler,\n                                   MinMaxScaler,\n                                   RobustScaler)\n\n\n\n\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.svm import OneClassSVM\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.covariance import EllipticEnvelope\nfrom sklearn.neighbors import LocalOutlierFactor\n\n!pip install umap-learn\nimport umap.umap_ as umap\nfrom sklearn.decomposition import PCA\nfrom imblearn.over_sampling import SMOTE,RandomOverSampler\n\n\n\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier, plot_importance, early_stopping\nfrom sklearn.ensemble import (AdaBoostClassifier,\n                              ExtraTreesClassifier,\n                              RandomForestClassifier,\n                              GradientBoostingClassifier)\n\n\n\n\nfrom sklearn.metrics import (accuracy_score,\n                             roc_auc_score,\n                             f1_score,\n                             recall_score,\n                             precision_score,\n                             recall_score,\n                             confusion_matrix)\n\n\n\n!pip install shap\nimport shap\n!pip install eli5\nimport eli5\nfrom eli5.sklearn import PermutationImportance\n\n!pip install vecstack\nfrom vecstack import stacking\n\n\n\n\n\nsns.set_style('white')\nmpl.rcParams['xtick.labelsize'] = 12\nmpl.rcParams['ytick.labelsize'] = 12\nmpl.rcParams['axes.spines.left'] = False\nmpl.rcParams['axes.spines.right'] = False\nmpl.rcParams['axes.spines.top'] = False\nmpl.rcParams['axes.spines.bottom'] = False\nplt.rcParams.update({'font.size':14})\nplt.rcParams['font.weight']= 'normal'","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:15:30.818683Z","iopub.execute_input":"2024-02-28T12:15:30.819145Z","iopub.status.idle":"2024-02-28T12:17:47.846675Z","shell.execute_reply.started":"2024-02-28T12:15:30.819104Z","shell.execute_reply":"2024-02-28T12:17:47.845194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"setup\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #F07B3F;\" id=\"setup\">6&nbsp;&nbsp;SETUP AND HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"def plot_eeg(levels, positions, axes, fig, ch_names=None, cmap='Spectral_r', cb_pos=(0.9, 0.1),\n             cb_width=0.04, cb_height=0.9, marker=None, marker_style=None, vmin=None, vmax=None, **kwargs):\n    \"\"\"\n    Function visulises processed EEG data in a simple way. Based on mne.viz.plot_topomap.\n\n\n    :param levels: numpy.array, shape (n_chan,)\n        data values to plot.\n    :param positions: numpy.array, shape (n_chan, 2)|instance of mne.Info\n        Location information for the data points(/channels). If an array, for each data point,\n        the x and y coordinates. If an Info object, it must contain only one data type and exactly\n        len(data) data channels, and the x/y coordinates will be inferred from the montage applied\n        to the Info object.\n    :param axes: matplotlib.axes.Axes\n        The axes to plot to.\n    :param fig: matplotlib.figure.Figure\n        The figure to create colorbar on.\n    :param ch_names: list | None\n        List of channel names. If None, channel names are not plotted.\n    :param cmap: matplotlib colormap | None\n        Colormap to use. If None, ‘Reds’ is used for all positive data, otherwise defaults to ‘RdBu_r’.\n        Default value is 'Spectral_r'\n    :param cb_pos: tuple/list of floats\n        Coordinates of color bar\n    :param cb_width: float\n        Width of colorbar\n    :param cb_height: float\n        Height of colorbar\n    :param marker: numpy.array of bool, shape (n_channels,) | None\n        Array indicating channel(s) to highlight with a distinct plotting style.\n        Array elements set to True will be plotted with the parameters given in mask_params.\n        Defaults to None, equivalent to an array of all False elements.\n    :param marker_style: dict | None\n        Additional plotting parameters for plotting significant sensors. Default (None) equals:\n        dict(marker='o', markerfacecolor='w', markeredgecolor='k', linewidth=0, markersize=4)\n    :param vmin, vmax: float | callable() | None\n        Lower and upper bounds of the colormap, in the same units as the data.\n        If vmin and vmax are both None, they are set at ± the maximum absolute value\n        of the data (yielding a colormap with midpoint at 0). If only one of vmin, vmax is None,\n        will use min(data) or max(data), respectively. If callable, should accept a NumPy array\n        of data and return a float.\n    :param kwargs:\n        any other parameter used in mne.viz.plot_topomap\n    :return im: matplotlib.image.AxesImage\n        The interpolated data.\n    :return cn: matplotlib.contour.ContourSet\n        The fieldlines.\n    \"\"\"\n    if 'mask' not in kwargs:\n        mask = np.ones(levels.shape[0], dtype='bool')\n    else:\n        mask = None\n    im, cm = mne.viz.plot_topomap(levels, positions, axes=axes, names=ch_names, cmap=cmap, mask=mask, mask_params=marker_style, show=False, **kwargs)\n\n    cbar_ax = fig.add_axes([cb_pos[0], cb_pos[1], cb_width, cb_height])\n    clb = axes.figure.colorbar(im, cax=cbar_ax)\n    return im, cm\n\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):\n    ret = {key:[] for key in x.columns}\n    \n    for pos in x.columns:\n        coeff = pywt.wavedec(x[pos], wavelet, mode=\"per\")\n        sigma = (1/0.6745) * maddest(coeff[-level])\n\n        uthresh = sigma * np.sqrt(2*np.log(len(x)))\n        coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n        ret[pos]=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return pd.DataFrame(ret)\n\ndef show_values_on_bars(axs, h_v=\"v\", space=0.4):\n    '''Plots the value at the end of the a seaborn barplot.\n    axs: the ax of the plot\n    h_v: weather or not the barplot is vertical/ horizontal'''\n    \n    def _show_on_single_plot(ax):\n        if h_v == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_height())\n                ax.text(_x, _y, format(value, ','), ha=\"center\") \n        elif h_v == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_width())\n                ax.text(_x, _y, format(value, ','), ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:20:36.901890Z","iopub.execute_input":"2024-02-28T12:20:36.904041Z","iopub.status.idle":"2024-02-28T12:20:36.927799Z","shell.execute_reply.started":"2024-02-28T12:20:36.903994Z","shell.execute_reply":"2024-02-28T12:20:36.926874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">6.2 LOAD THE DATA</h3>","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/hms-harmful-brain-activity-classification\"\n\n# Define the paths to the batches of train and test data respectively\nTRAIN_CSV_PATH = os.path.join(DATA_DIR, \"train.csv\")\nTEST_CSV_PATH  = os.path.join(DATA_DIR, \"test.csv\")\nSS_CSV_PATH  = os.path.join(DATA_DIR, \"sample_submission.csv\")\n\n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\nprint(\"\\n\\n... LOAD TRAIN DATAFRAME FROM CSV FILE ...\\n\")\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ndisplay(train_df)\n\nprint(\"\\n\\n... LOAD TEST DATAFRAME FROM CSV FILE ...\\n\")\ntest_df = pd.read_csv(TEST_CSV_PATH)\ndisplay(test_df)\n\nprint(\"\\n\\n... LOAD SAMPLE SUBMISSION DATAFRAME FROM CSV FILE ...\\n\")\nss_df = pd.read_csv(SS_CSV_PATH)\ndisplay(ss_df)\n\n\nprint(\"\\n\\n\\n... BASIC DATA SETUP FINISHED ...\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:21:54.812842Z","iopub.execute_input":"2024-02-28T12:21:54.813311Z","iopub.status.idle":"2024-02-28T12:21:55.177475Z","shell.execute_reply.started":"2024-02-28T12:21:54.813280Z","shell.execute_reply":"2024-02-28T12:21:55.176236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(color_class.BOLD_YELLOW + 'Shape of The Dataset:'+ color_class.END + color_class.BOLD_RED + ' {}'.format(train_df.shape) +2*'\\n' + 'Note:  There is no missing value on data'.format(train_df.isnull().any().sum())  +\n     '\\n')\n\nprint(color_class.BOLD_RED + 'Null/Missing values are as follows: :'+ color_class.END + '\\n')\nprint(train_df.isnull().sum())\nprint(color_class.BOLD_YELLOW + 'Since there are no null/missing values in any column, the process will proceed.'+ color_class.END)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:26:08.049456Z","iopub.execute_input":"2024-02-28T12:26:08.049946Z","iopub.status.idle":"2024-02-28T12:26:08.088501Z","shell.execute_reply.started":"2024-02-28T12:26:08.049888Z","shell.execute_reply":"2024-02-28T12:26:08.087367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"data_exploration\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #F07B3F;\" id=\"data_exploration\">7&nbsp;&nbsp;DATA EXPLORATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\" style=\"color: #FD3792;\">&#10514;</a></h1>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.1 PATIENT ID</h3>","metadata":{}},{"cell_type":"code","source":"print(f\"Number of patients:  {train_df['patient_id'].nunique()}\")\nprint(color_class.BOLD_RED + 'Lets check the duplicate ID from the patient'+ color_class.END)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:26:23.769026Z","iopub.execute_input":"2024-02-28T12:26:23.769542Z","iopub.status.idle":"2024-02-28T12:26:23.782658Z","shell.execute_reply.started":"2024-02-28T12:26:23.769501Z","shell.execute_reply":"2024-02-28T12:26:23.781436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duplicate_counts = train_df['patient_id'].value_counts()\ntop_10_duplicate_counts = duplicate_counts.head(10)\n\ncolors = cm.plasma(top_10_duplicate_counts / float(max(top_10_duplicate_counts)))\n\nplt.figure(figsize=(10, 6))\nbars = top_10_duplicate_counts.plot(kind='barh', color=colors)\nplt.title('Top 10 Duplicate Counts for Patient IDs')\nplt.xlabel('Count')\nplt.ylabel('Patient ID')\n\n# Tambahkan nilai rinci di setiap bar\nfor bar, count in zip(bars.patches, top_10_duplicate_counts.values):\n    plt.text(bar.get_width(), bar.get_y() + bar.get_height() / 2, f'{count}', \n             va='center', ha='left', color='black')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T03:42:28.281317Z","iopub.execute_input":"2024-02-28T03:42:28.281645Z","iopub.status.idle":"2024-02-28T03:42:28.611364Z","shell.execute_reply.started":"2024-02-28T03:42:28.281621Z","shell.execute_reply":"2024-02-28T03:42:28.610260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.2 EXPERT CONSESUS</h3>","metadata":{}},{"cell_type":"code","source":"expert_consensus = train_df['expert_consensus'].value_counts()\n\ncolors = cm.plasma(expert_consensus / float(max(expert_consensus)))\n\nplt.figure(figsize=(10, 6))\nbars = expert_consensus.plot(kind='barh', color=colors)\nplt.title('Expert Consensus Counts In Data')\nplt.xlabel('Count')\nplt.ylabel('Expert Consensus')\n\nfor bar, count in zip(bars.patches, expert_consensus.values):\n    plt.text(bar.get_width(), bar.get_y() + bar.get_height() / 2, f'{count}', \n             va='center', ha='left', color='black')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T03:42:33.713462Z","iopub.execute_input":"2024-02-28T03:42:33.713796Z","iopub.status.idle":"2024-02-28T03:42:34.020858Z","shell.execute_reply.started":"2024-02-28T03:42:33.713768Z","shell.execute_reply":"2024-02-28T03:42:34.019881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3 DESCRIPTIVE STATISTICS</h3>","metadata":{}},{"cell_type":"code","source":"# Statistika deskriptif\nstats_df = train_df.describe().T.reset_index().rename(columns = {'index':'Features'})\nstats_df['count'] = stats_df['count'].astype(int)\n\n# Visualisasi\nstyle = stats_df.style.set_table_attributes(\"style='display:inline'\").\\\n                                bar(subset = ['mean', 'std', 'min', '25%','50%','75%', 'max'],axis = 1 ,color = '#FB8500')\\\n                                .format({\n                                         'mean':\"{:20,.3f}\",\n                                         'std':\"{:20,.3f}\",\n                                         'min':\"{:20,.3f}\",\n                                         '25%':\"{:20,.3f}\",\n                                         '50%':\"{:20,.3f}\",\n                                         '75%':\"{:20,.3f}\",\n                                         'max':\"{:20,.3f}\"})\\\n                                .format({\"Features\": lambda x:  x.upper()},\n                                 )\\\n                                .set_properties(**{'background-color': 'white',\n                                    'color': 'black'})\nprint(color_class.BOLD_RED + '''\\n Descriptive Statistics from Dataset \\\n\\n''' + color_class.END)\n\ndisplay_html(style._repr_html_(), raw=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T03:42:45.321503Z","iopub.execute_input":"2024-02-28T03:42:45.321900Z","iopub.status.idle":"2024-02-28T03:42:45.442580Z","shell.execute_reply.started":"2024-02-28T03:42:45.321866Z","shell.execute_reply":"2024-02-28T03:42:45.441676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nplt.hist(train_df['eeg_label_offset_seconds'], bins='auto', log=True)\nplt.xlabel('Offset Seconds')\nplt.ylabel('Frequency (Log Scale)')\nplt.title('Histogram of Offset Seconds (Log Scale)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T03:42:56.259239Z","iopub.execute_input":"2024-02-28T03:42:56.259571Z","iopub.status.idle":"2024-02-28T03:42:58.466041Z","shell.execute_reply.started":"2024-02-28T03:42:56.259543Z","shell.execute_reply":"2024-02-28T03:42:58.465189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3 EEG TRAIN</h3>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3.1 EEG TRAIN SIZE</h3>","metadata":{}},{"cell_type":"code","source":"print(color_class.BOLD_RED + 'Lets see how many files and sizes that we have in this eeg train folder'+ color_class.END + '\\n')\nprint(\"Train eeg files: \")\n! ls /kaggle/input/hms-harmful-brain-activity-classification/train_eegs | wc -l\nprint(\"Size of train eegs\")\n! du -sh /kaggle/input/hms-harmful-brain-activity-classification/train_eegs","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:26:42.931090Z","iopub.execute_input":"2024-02-28T12:26:42.931531Z","iopub.status.idle":"2024-02-28T12:27:25.209274Z","shell.execute_reply.started":"2024-02-28T12:26:42.931500Z","shell.execute_reply":"2024-02-28T12:27:25.208078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3.2 EXAMPLE OF OUR DATA</h3>","metadata":{}},{"cell_type":"code","source":"print(color_class.BOLD_RED + 'Lets take a look at  examples of our file'+ color_class.END + '\\n')\nsample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1001487592.parquet\")\nsample_train_eeg","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:27:25.212432Z","iopub.execute_input":"2024-02-28T12:27:25.213751Z","iopub.status.idle":"2024-02-28T12:27:25.414157Z","shell.execute_reply.started":"2024-02-28T12:27:25.213683Z","shell.execute_reply":"2024-02-28T12:27:25.413233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://upload.wikimedia.org/wikipedia/commons/thumb/f/fb/EEG_10-10_system_with_additional_information.svg/1280px-EEG_10-10_system_with_additional_information.svg.png\" alt=\"670px-International-10-20-system-for-EEG-MCN-svg\" border=\"0\">\n\n<br>Each electrode placement site has a letter to identify the lobe, or area of the brain it is reading from: pre-frontal (Fp), frontal (F), temporal (T), parietal (P), occipital (O), and central (C). Note that there is no \"central lobe\"; due to their placement, and depending on the individual, the \"C\" electrodes can exhibit/represent EEG activity more typical of frontal, temporal, and some parietal-occipital activity, and are always utilized in polysomnography sleep studies for the purpose of determining stages of sleep.\n\nThere are also (Z) sites: A \"Z\" (zero) refers to an electrode placed on the midline sagittal plane of the skull, (FpZ, Fz, Cz, Oz) and is present mostly for reference/measurement points. These electrodes will not necessarily reflect or amplify lateral hemispheric cortical activity as they are placed over the corpus callosum, and do not represent either hemisphere adequately. \"Z\" electrodes are often utilized as 'grounds' or 'references,' especially in polysomnography sleep studies, and diagnostic/clinical EEG montages meant to represent/diagnose epileptiform seizure activity, or possible clinical brain death. Note that the required number of EEG electrodes, and their careful, measured placement, increases with each clinical requirement and modality.\n\nEven-numbered electrodes (2,4,6,8) refer to electrode placement on the right side of the head, whereas odd numbers (1,3,5,7) refer to those on the left; this applies to both EEG and EOG (electrooculogram measurements of eyes) electrodes, as well as ECG (electrocardiography measurements of the heart) electrode placement. Chin, or EMG (electromyogram) electrodes are more commonly just referred to with \"right,\" \"left,\" and \"reference,\" or \"common,\" as there are usually only three placed, and they can be differentially referenced from the EEG and EOG reference sites.\n\nThe \"A\" (sometimes referred to as \"M\" for mastoid process) refers to the prominent bone process usually found just behind the outer ear (less prominent in children and some adults). In basic polysomnography, F3, F4, Fz, Cz, C3, C4, O1, O2, A1, A2 (M1, M2), are used. Cz and Fz are 'ground' or 'common' reference points for all EEG and EOG electrodes, and A1-A2 are used for contralateral referencing of all EEG electrodes. This EEG montage may be extended to utilize T3-T4, P3-P4, as well as others, if an extended or \"seizure montage\" is called for.</br>\n\n<br>Specific anatomical landmarks are used for the essential measuring and positioning of the EEG electrodes. These are found with a tape measure, and often marked with a grease pencil, or \"China marker.\"\n\n* Nasion to Inion: the nasion is the distinctly depressed area between the eyes, just above the bridge of the nose, and the inion, is the crest point of back of the skull, often indicated by a bump (the prominent occipital ridge, can usually be located with mild palpation). Marks for the Z electrodes are made between these points along the midline, at intervals of 10%, 20%, 20%, 20%, 20% and 10%.\n\n* Preauricular to preauricular (or tragus to tragus: the tragus refers to the small portion of cartilage projecting anteriorly to the pinna). The preauricular point is in front of each ear, and can be more easily located with mild palpation, and if necessary, requesting patient to open mouth slightly. The T3, C3, Cz, C4, and T4 electrodes are placed at marks made at intervals of 10%, 20%, 20%, 20%, 20% and 10%, respectively, measured across the top of the head.\n\n* Skull circumference is measured just above the ears (T3 and T4), just above the bridge of the nose (at Fpz), and just above the occipital point (at Oz). The Fp2, F8, T4, T6, and O2 electrodes are placed at intervals of 5%, 10%, 10%, 10%, 10%, and 5%, respectively, measured above the right ear, from front (Fpz) to back (Oz). The same is done for the odd-numbered electrodes on the left side, to complete the full circumference.\n\n* Measurement methods for placement of the F3, F4, P3, and P4 points differ. If measured front-to-back (Fp1-F3-C3-P3-O1 and Fp2-F4-C4-P4-O2 montages), they can be 25% \"up\" from the front and back points (Fp1, Fp2, O1, and O2). If measured side-to-side (F7-F3-Fz-F4-F8 and T5-P3-Pz-P4-T6 montages), they can be 25% \"up\" from the side points (F7, F8, T5, and T6). If measured diagonally, from Nasion to Inion through the C3 and C4 points, they will be 20% in front of and behind the C3 and C4 points. Each of these measurement methods results in different nominal electrode placements.\n\nWhen placing the A (or M) electrodes, palpation is often necessary to determine the most pronounced point of the mastoid process behind either ear; failure to do so, and to place the reference electrodes too low (posterior to the ear pinna, proximal to the throat) may result in \"EKG artifact\" in the EEGs and EOGs, due to artifact from the carotid arteries. EKG artifact can be reduced with post-filtering of signals, or by \"jumping\" (co-referencing) of A/M reference electrodes, if replacement of reference electrodes is not possible, ameliorative, or if other clinical considerations prevent otherwise good placement (such as congenital malformation, or post-surgical considerations such as Cochlear Implants).</br>\n\n<br>When recording a more detailed EEG with more electrodes, extra electrodes are added using the 10% division, which fills in intermediate sites halfway between those of the existing 10–20 system. This new electrode-naming-system is more complicated giving rise to the Modified Combinatorial Nomenclature (MCN). This MCN system uses 1, 3, 5, 7, 9 for the left hemisphere which represents 10%, 20%, 30%, 40%, 50% of the inion-to-nasion distance respectively. The introduction of extra letter codes allows the naming of intermediate electrode sites. Note that these new letter codes do not necessarily refer to an area on the underlying cerebral cortex.</br>\n\nThe new letter codes of the MCN for intermediate electrode places are:\n\n1. AF – between Fp and F\n2. FC – between F and C\n3. FT – between F and T\n4. CP – between C and P\n5. TP – between T and P\n6. PO – between P and O\n\n<br>Also, the MCN system renames four electrodes of the 10–20 system:</br>\n\n1. T3 is now T7\n2. T4 is now T8\n3. T5 is now P7\n4. T6 is now P8","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3.3 CHECK THE COLUMN OF OUR DATA</h3>","metadata":{}},{"cell_type":"code","source":"folder_path = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs\"\ncolumns = set()\n\nfor filename in os.listdir(folder_path):\n    if filename.endswith(\".parquet\"):\n        file_path = os.path.join(folder_path, filename)\n        \n       \n        df = pd.read_parquet(file_path)\n        \n      \n        columns.update(df.columns)\n\nprint(f\"The unique columns present in all file: {(columns)}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-28T03:44:05.664071Z","iopub.execute_input":"2024-02-28T03:44:05.664425Z","iopub.status.idle":"2024-02-28T03:50:58.476009Z","shell.execute_reply.started":"2024-02-28T03:44:05.664395Z","shell.execute_reply":"2024-02-28T03:50:58.474577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It can be seen in the output that some columns still follow the old naming rule.\n\n<br>\n\n<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">but do we need to rename these columns? </h3>\n\nThe renaming of columns is based on the analytical needs of each researcher. For this session, I did not change those columns to follow the new rules","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.3.4 Visualization of EEG Train</h3>","metadata":{}},{"cell_type":"code","source":"# Set the color palette to Viridis\nsns.set_palette('rocket_r')\n\nfig, ax = plt.subplots(20, figsize=(10, 100))\n\n# Generate a line plot for each column in the DataFrame\nfor i, column in enumerate(sample_train_eeg.columns):\n    ax[i].plot(sample_train_eeg.index, sample_train_eeg[column], label=column)\n    ax[i].grid(True)\n    ax[i].set_title(str(column))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:44:15.456956Z","iopub.execute_input":"2024-01-15T12:44:15.457555Z","iopub.status.idle":"2024-01-15T12:44:24.205576Z","shell.execute_reply.started":"2024-01-15T12:44:15.457501Z","shell.execute_reply":"2024-01-15T12:44:24.204020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<b>In the visualization, it can be observed that the obtained EEG signals are still mixed with physiological artifacts. We will proceed with the denoising process for modeling.</b>\n\nLet's visualize these signals with mapping on our brain.","metadata":{}},{"cell_type":"markdown","source":"In this section i use this package\n\n<img src=\"https://mne.tools/stable/_images/mne_logo.svg\" alt=\"670px-International-10-20-system-for-EEG-MCN-svg\" border=\"0\">\n\nOpen-source Python package for exploring, visualizing, and analyzing human neurophysiological data: MEG, EEG, sEEG, ECoG, NIRS, and more.\n\nfor further documentation and information you can acess link :\n[MNE_PACKAGE](http://mne.tools/stable/index.html)","metadata":{}},{"cell_type":"code","source":"# import mne\n# from mne.channels import make_standard_montage\n# from mne.io import RawArray","metadata":{"execution":{"iopub.status.busy":"2024-01-15T09:40:19.728127Z","iopub.execute_input":"2024-01-15T09:40:19.728820Z","iopub.status.idle":"2024-01-15T09:40:19.736168Z","shell.execute_reply.started":"2024-01-15T09:40:19.728764Z","shell.execute_reply":"2024-01-15T09:40:19.734357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#In this code, I am creating custom coordinates to map the signals on the brain.\nchs = {'Fp1': [-0.03, 0.08],\n       'Fp2': [0.03, 0.08],\n       'F7': [-0.073, 0.047],\n       'F3': [-0.04, 0.041],\n       'Fz': [0, 0.038],\n       'F4': [0.04, 0.041],\n       'F8': [0.073, 0.047],\n       'T3': [-0.085, 0],\n       'C3': [-0.045, 0],\n       'Cz': [0, 0],\n       'C4': [0.045, 0],\n       'T4': [0.085, 0],\n       'T5': [-0.073, -0.047],\n       'P3': [-0.04, -0.041],\n       'Pz': [0, -0.038],\n       'P4': [0.04, -0.041],\n       'T6': [0.07, -0.047],\n       'O1': [-0.03, -0.08],\n       'O2': [0.03, -0.08],\n       'Oz' :[0, -0.08] }\n\nchannels = pd.DataFrame(chs).transpose()\nchannels","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:44:46.616571Z","iopub.execute_input":"2024-01-15T12:44:46.617072Z","iopub.status.idle":"2024-01-15T12:44:46.645190Z","shell.execute_reply.started":"2024-01-15T12:44:46.617034Z","shell.execute_reply":"2024-01-15T12:44:46.643871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Since it is required in three dimensions, I added a value of 0 for the third coordinate of each electrode\nfor key in chs.keys():\n    chs[key]+=[0]\nchs","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:44:52.422603Z","iopub.execute_input":"2024-01-15T12:44:52.423099Z","iopub.status.idle":"2024-01-15T12:44:52.434263Z","shell.execute_reply.started":"2024-01-15T12:44:52.423058Z","shell.execute_reply":"2024-01-15T12:44:52.432918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mont = mne.channels.make_dig_montage(chs)\nmont.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:45:00.015795Z","iopub.execute_input":"2024-01-15T12:45:00.016263Z","iopub.status.idle":"2024-01-15T12:45:00.385105Z","shell.execute_reply.started":"2024-01-15T12:45:00.016227Z","shell.execute_reply":"2024-01-15T12:45:00.383479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is the shape we will use for mapping EEG signals on the brain.","metadata":{}},{"cell_type":"code","source":"# In this code i take mean value from eeg data train for every electrode position\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\nmean_df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:47:47.568797Z","iopub.execute_input":"2024-01-15T12:47:47.569316Z","iopub.status.idle":"2024-01-15T12:47:47.587406Z","shell.execute_reply.started":"2024-01-15T12:47:47.569277Z","shell.execute_reply":"2024-01-15T12:47:47.586141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Merged data\nmerged_df = channels.join(mean_df.set_index('channels'), how='left').fillna(0)\nmerged_df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:47:50.674371Z","iopub.execute_input":"2024-01-15T12:47:50.675722Z","iopub.status.idle":"2024-01-15T12:47:50.697092Z","shell.execute_reply.started":"2024-01-15T12:47:50.675644Z","shell.execute_reply":"2024-01-15T12:47:50.695808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Merged data\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1001487592')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:48:41.736486Z","iopub.execute_input":"2024-01-15T12:48:41.737883Z","iopub.status.idle":"2024-01-15T12:48:42.251037Z","shell.execute_reply.started":"2024-01-15T12:48:41.737830Z","shell.execute_reply":"2024-01-15T12:48:42.249697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We can see from the visualization that the frontal (F4) and center right (C4) parts of the brain have the most negative average EEG signals, while the opitcal (O1) part has the most positive average EEG signal.**\n\n<br> Lets take a look for another example","metadata":{}},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1001369401.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\nmerged_df = channels.join(mean_df.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1001369401')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:48:56.253887Z","iopub.execute_input":"2024-01-15T12:48:56.254878Z","iopub.status.idle":"2024-01-15T12:48:56.768511Z","shell.execute_reply.started":"2024-01-15T12:48:56.254831Z","shell.execute_reply":"2024-01-15T12:48:56.767082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now we want to see example mapping for each expert consensus**","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">LPD Mapping Example </h3>","metadata":{}},{"cell_type":"code","source":"lpd_train = train_df[train_df['expert_consensus'] == 'LPD']\nlpd_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T10:14:43.675862Z","iopub.execute_input":"2024-01-15T10:14:43.676407Z","iopub.status.idle":"2024-01-15T10:14:43.774343Z","shell.execute_reply.started":"2024-01-15T10:14:43.676371Z","shell.execute_reply":"2024-01-15T10:14:43.772822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/736446371.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 736446371 with LPD Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:50:25.515035Z","iopub.execute_input":"2024-01-15T12:50:25.516056Z","iopub.status.idle":"2024-01-15T12:50:26.046725Z","shell.execute_reply.started":"2024-01-15T12:50:25.516007Z","shell.execute_reply":"2024-01-15T12:50:26.045291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1306668185.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1306668185 with LPD Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:50:58.598761Z","iopub.execute_input":"2024-01-15T12:50:58.599246Z","iopub.status.idle":"2024-01-15T12:50:59.171916Z","shell.execute_reply.started":"2024-01-15T12:50:58.599210Z","shell.execute_reply":"2024-01-15T12:50:59.170359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">LRDA Mapping Example </h3>","metadata":{}},{"cell_type":"code","source":"lrda_train = train_df[train_df['expert_consensus'] == 'LRDA']\nlrda_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:08:16.276092Z","iopub.execute_input":"2024-01-15T13:08:16.277469Z","iopub.status.idle":"2024-01-15T13:08:16.329035Z","shell.execute_reply.started":"2024-01-15T13:08:16.277398Z","shell.execute_reply":"2024-01-15T13:08:16.327655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/722738444.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 722738444 with LRDA Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:52:22.455336Z","iopub.execute_input":"2024-01-15T12:52:22.455917Z","iopub.status.idle":"2024-01-15T12:52:23.003597Z","shell.execute_reply.started":"2024-01-15T12:52:22.455872Z","shell.execute_reply":"2024-01-15T12:52:23.002037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/351917269.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 351917269 with LRDA Expert Consensus')\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:54:15.165414Z","iopub.execute_input":"2024-01-15T12:54:15.166596Z","iopub.status.idle":"2024-01-15T12:54:15.722572Z","shell.execute_reply.started":"2024-01-15T12:54:15.166545Z","shell.execute_reply":"2024-01-15T12:54:15.721203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/387987538.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 387987538 with LRDA Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:07:45.376504Z","iopub.execute_input":"2024-01-15T13:07:45.377058Z","iopub.status.idle":"2024-01-15T13:07:45.924335Z","shell.execute_reply.started":"2024-01-15T13:07:45.377018Z","shell.execute_reply":"2024-01-15T13:07:45.923375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">GPD Mapping Example </h3>","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:56:56.695145Z","iopub.execute_input":"2024-01-15T12:56:56.695726Z","iopub.status.idle":"2024-01-15T12:56:56.704563Z","shell.execute_reply.started":"2024-01-15T12:56:56.695684Z","shell.execute_reply":"2024-01-15T12:56:56.702699Z"}}},{"cell_type":"code","source":"gpd_train = train_df[train_df['expert_consensus'] == 'GPD']\ngpd_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T12:57:28.466165Z","iopub.execute_input":"2024-01-15T12:57:28.466711Z","iopub.status.idle":"2024-01-15T12:57:28.518932Z","shell.execute_reply.started":"2024-01-15T12:57:28.466659Z","shell.execute_reply":"2024-01-15T12:57:28.517705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2277392603.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 2277392603 with GPD Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:04:11.515102Z","iopub.execute_input":"2024-01-15T13:04:11.515672Z","iopub.status.idle":"2024-01-15T13:04:12.055400Z","shell.execute_reply.started":"2024-01-15T13:04:11.515615Z","shell.execute_reply":"2024-01-15T13:04:12.054198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/531742289.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 531742289 with GPD Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:04:49.744896Z","iopub.execute_input":"2024-01-15T13:04:49.745733Z","iopub.status.idle":"2024-01-15T13:04:50.290007Z","shell.execute_reply.started":"2024-01-15T13:04:49.745687Z","shell.execute_reply":"2024-01-15T13:04:50.288999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/675867705.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 675867705 with GPD Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:05:43.567747Z","iopub.execute_input":"2024-01-15T13:05:43.568226Z","iopub.status.idle":"2024-01-15T13:05:44.135676Z","shell.execute_reply.started":"2024-01-15T13:05:43.568192Z","shell.execute_reply":"2024-01-15T13:05:44.134184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">GRDA Mapping Example </h3>","metadata":{}},{"cell_type":"code","source":"grda_train = train_df[train_df['expert_consensus'] == 'GRDA']\ngrda_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:12:47.219283Z","iopub.execute_input":"2024-01-15T13:12:47.219863Z","iopub.status.idle":"2024-01-15T13:12:47.276452Z","shell.execute_reply.started":"2024-01-15T13:12:47.219820Z","shell.execute_reply":"2024-01-15T13:12:47.275119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2578018731.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 2578018731 with GRDA Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:13:33.999926Z","iopub.execute_input":"2024-01-15T13:13:34.000446Z","iopub.status.idle":"2024-01-15T13:13:34.545041Z","shell.execute_reply.started":"2024-01-15T13:13:34.000406Z","shell.execute_reply":"2024-01-15T13:13:34.543031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/8071080.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 8071080 with GRDA Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:14:10.300325Z","iopub.execute_input":"2024-01-15T13:14:10.301143Z","iopub.status.idle":"2024-01-15T13:14:10.887532Z","shell.execute_reply.started":"2024-01-15T13:14:10.301078Z","shell.execute_reply":"2024-01-15T13:14:10.886116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1142142950.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1142142950 with GRDA Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:14:44.530086Z","iopub.execute_input":"2024-01-15T13:14:44.530598Z","iopub.status.idle":"2024-01-15T13:14:45.107196Z","shell.execute_reply.started":"2024-01-15T13:14:44.530559Z","shell.execute_reply":"2024-01-15T13:14:45.105785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">Seizure Mapping Example </h3>","metadata":{}},{"cell_type":"code","source":"seizure_train = train_df[train_df['expert_consensus'] == 'Seizure']\nseizure_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:29:42.478786Z","iopub.execute_input":"2024-01-15T13:29:42.479384Z","iopub.status.idle":"2024-01-15T13:29:42.531481Z","shell.execute_reply.started":"2024-01-15T13:29:42.479334Z","shell.execute_reply":"2024-01-15T13:29:42.530144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1628180742.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1628180742 with Seizure Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:33:54.283645Z","iopub.execute_input":"2024-01-15T13:33:54.284221Z","iopub.status.idle":"2024-01-15T13:33:54.868039Z","shell.execute_reply.started":"2024-01-15T13:33:54.284182Z","shell.execute_reply":"2024-01-15T13:33:54.866691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3349371726.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 3349371726 with Seizure Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:34:35.083692Z","iopub.execute_input":"2024-01-15T13:34:35.084188Z","iopub.status.idle":"2024-01-15T13:34:35.628146Z","shell.execute_reply.started":"2024-01-15T13:34:35.084150Z","shell.execute_reply":"2024-01-15T13:34:35.627155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/641615525.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 641615525 with Seizure Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:35:07.643406Z","iopub.execute_input":"2024-01-15T13:35:07.644765Z","iopub.status.idle":"2024-01-15T13:35:08.172527Z","shell.execute_reply.started":"2024-01-15T13:35:07.644712Z","shell.execute_reply":"2024-01-15T13:35:08.171477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 16px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">Other Mapping Example </h3>","metadata":{}},{"cell_type":"code","source":"other_train = train_df[train_df['expert_consensus'] == 'Other']\nother_train","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:39:58.020490Z","iopub.execute_input":"2024-01-15T13:39:58.021821Z","iopub.status.idle":"2024-01-15T13:39:58.076306Z","shell.execute_reply.started":"2024-01-15T13:39:58.021761Z","shell.execute_reply":"2024-01-15T13:39:58.075116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1202099836.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 1202099836 with Other Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:40:37.320860Z","iopub.execute_input":"2024-01-15T13:40:37.321371Z","iopub.status.idle":"2024-01-15T13:40:37.864264Z","shell.execute_reply.started":"2024-01-15T13:40:37.321334Z","shell.execute_reply":"2024-01-15T13:40:37.862885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3037445252.parquet\")\nmean_df = sample_train_eeg.mean().reset_index()\nmean_df.columns = ['channels', 'mean']\nmean_df = mean_df[mean_df['channels'] != 'EKG']\ntest = mean_df\nmerged_df = channels.join(test.set_index('channels'), how='left').fillna(0)\n# extract power for one main.disorder and one band\nmerged_df = merged_df['mean']\n\n# make sure that channels are in correct order\nassert (merged_df.index == channels.index).all()\n# plot\nfig, ax = plt.subplots()\nplot_eeg(merged_df, channels.to_numpy(), ax, fig, vmin = None, marker_style={'markersize':4, 'markerfacecolor':'black'})\nplt.title('EEG Signal Mapping From id 3037445252 with Other Expert Consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:41:07.312555Z","iopub.execute_input":"2024-01-15T13:41:07.313714Z","iopub.status.idle":"2024-01-15T13:41:07.842229Z","shell.execute_reply.started":"2024-01-15T13:41:07.313660Z","shell.execute_reply":"2024-01-15T13:41:07.841294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.4 SPECTOGRAM TRAIN</h3>","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:44:42.357188Z","iopub.execute_input":"2024-01-15T13:44:42.357812Z","iopub.status.idle":"2024-01-15T13:44:42.366424Z","shell.execute_reply.started":"2024-01-15T13:44:42.357768Z","shell.execute_reply":"2024-01-15T13:44:42.364650Z"}}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.4.1 SPECTOGRAM TRAIN SIZE</h3>","metadata":{}},{"cell_type":"code","source":"print(color_class.BOLD_RED + 'Lets see how many files and sizes that we have in this spectogram train folder'+ color_class.END + '\\n')\nprint(\"Train spectogram files: \")\n! ls /kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms | wc -l\nprint(\"Size of train spectogram\")\n! du -sh /kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:47:37.163822Z","iopub.execute_input":"2024-01-15T13:47:37.164383Z","iopub.status.idle":"2024-01-15T13:47:47.776247Z","shell.execute_reply.started":"2024-01-15T13:47:37.164343Z","shell.execute_reply":"2024-01-15T13:47:47.775058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 18px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #F07B3F; background-color: #ffffff;\">7.4.2 EXAMPLE OF OUR DATA</h3>","metadata":{}},{"cell_type":"code","source":"print(color_class.BOLD_RED + 'Lets take a look at  examples of our file'+ color_class.END + '\\n')\nsample_train_spec = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet\")\nsample_train_spec","metadata":{"execution":{"iopub.status.busy":"2024-01-15T13:49:36.633119Z","iopub.execute_input":"2024-01-15T13:49:36.634289Z","iopub.status.idle":"2024-01-15T13:49:36.740020Z","shell.execute_reply.started":"2024-01-15T13:49:36.634231Z","shell.execute_reply":"2024-01-15T13:49:36.738878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We can see that in train_spectograms there is missing values**","metadata":{}},{"cell_type":"code","source":"def plot_spectrogram(spectrogram_path):\n    sample_spect = pd.read_parquet(spectrogram_path)\n    \n    split_spect = {\n        \"LL\": sample_spect.filter(regex='^LL', axis=1),\n        \"RL\": sample_spect.filter(regex='^RL', axis=1),\n        \"RP\": sample_spect.filter(regex='^RP', axis=1),\n        \"LP\": sample_spect.filter(regex='^LP', axis=1),\n    }\n    \n    fig, axes = plt.subplots(nrows=2, ncols=2, figsize=(15, 12))\n    axes = axes.flatten()\n    label_interval = 5\n    for i, split_name in enumerate(split_spect.keys()):\n        ax = axes[i]\n        img = ax.imshow(np.log(split_spect[split_name]).T, cmap='viridis', aspect='auto', origin='lower')\n        cbar = fig.colorbar(img, ax=ax)\n        cbar.set_label('Log(Value)')\n        ax.set_title(split_name)\n        ax.set_ylabel(\"Frequency (Hz)\")\n        ax.set_xlabel(\"Time\")\n\n        ax.set_yticks(np.arange(len(split_spect[split_name].columns)))\n        ax.set_yticklabels([column_name[3:] for column_name in split_spect[split_name].columns])\n        frequencies = [column_name[3:] for column_name in split_spect[split_name].columns]\n        ax.set_yticks(np.arange(0, len(split_spect[split_name].columns), label_interval))\n        ax.set_yticklabels(frequencies[::label_interval])\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T14:00:22.193131Z","iopub.execute_input":"2024-01-15T14:00:22.193753Z","iopub.status.idle":"2024-01-15T14:00:22.210395Z","shell.execute_reply.started":"2024-01-15T14:00:22.193703Z","shell.execute_reply":"2024-01-15T14:00:22.208787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spectrogram(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000189855.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-01-15T14:00:49.955100Z","iopub.execute_input":"2024-01-15T14:00:49.955944Z","iopub.status.idle":"2024-01-15T14:00:54.794409Z","shell.execute_reply.started":"2024-01-15T14:00:49.955884Z","shell.execute_reply":"2024-01-15T14:00:54.792811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, gc\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-01-25T14:27:26.825628Z","iopub.execute_input":"2024-01-25T14:27:26.826275Z","iopub.status.idle":"2024-01-25T14:27:26.833581Z","shell.execute_reply.started":"2024-01-25T14:27:26.826218Z","shell.execute_reply":"2024-01-25T14:27:26.832033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create Non Overlapping data","metadata":{}},{"cell_type":"code","source":"# Creating a Unique EEG Segment per eeg_id:\n# The code groups (groupby) the EEG data (df) by eeg_id. Each eeg_id represents a different EEG recording.\n# It then picks the first spectrogram_id and the earliest (min) spectrogram_label_offset_seconds for each eeg_id. This helps in identifying the starting point of each EEG segment.\n# The resulting DataFrame train has columns spec_id (first spectrogram_id) and min (earliest spectrogram_label_offset_seconds).\nTARGETS = train_df.columns[-6:]\ntrain = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\n\n# Finding the Latest Point in Each EEG Segment:\n# The code again groups the data by eeg_id and finds the latest (max) spectrogram_label_offset_seconds for each segment.\n# This max value is added to the train DataFrame, representing the end point of each EEG segment.\ntmp = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\n\ntmp = train_df.groupby('eeg_id')[['patient_id']].agg('first') # The code adds the patient_id for each eeg_id to the train DataFrame. This links each EEG segment to a specific patient.\ntrain['patient_id'] = tmp\n\n\ntmp = train_df.groupby('eeg_id')[TARGETS].agg('sum') # The code sums up the target variable counts (like votes for seizure, LPD, etc.) for each eeg_id.\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values # It then normalizes these counts so that they sum up to 1. This step converts the counts into probabilities, which is a common practice in classification tasks.\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = train_df.groupby('eeg_id')[['expert_consensus']].agg('first') # For each eeg_id, the code includes the expert_consensus on the EEG segment's classification.\ntrain['target'] = tmp\n\ntrain = train.reset_index() # This makes eeg_id a regular column, making the DataFrame easier to work with.\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:27:25.415645Z","iopub.execute_input":"2024-02-28T12:27:25.416100Z","iopub.status.idle":"2024-02-28T12:27:25.531723Z","shell.execute_reply.started":"2024-02-28T12:27:25.416064Z","shell.execute_reply":"2024-02-28T12:27:25.530540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_SPEC_FILES = False # If READ_SPEC_FILES is False, the code reads the combined file instead of individual files.\nFEATURE_ENGINEER = True","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:27:58.498936Z","iopub.execute_input":"2024-02-28T12:27:58.499406Z","iopub.status.idle":"2024-02-28T12:27:58.504640Z","shell.execute_reply.started":"2024-02-28T12:27:58.499366Z","shell.execute_reply":"2024-02-28T12:27:58.503353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:28:03.296670Z","iopub.execute_input":"2024-02-28T12:28:03.297101Z","iopub.status.idle":"2024-02-28T12:29:26.130674Z","shell.execute_reply.started":"2024-02-28T12:28:03.297067Z","shell.execute_reply":"2024-02-28T12:29:26.128462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ENGINEER FEATURES\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# The code generates features from the spectrogram data for use in a model \n# The features are derived by calculating the mean, minimum, maximum, and standard deviation values over time for each of the 400 spectrogram frequencies.\n# Two types of windows are used for these calculations:\n# A 10-minute window (_mean_10m, _min_10m, _max_10m, _std_10m).\n# A 20-second window (_mean_20s, _min_20s, _max_20s, _std_20s).\n# This process results in 2400 features (400 features × 6 calculations) for each EEG ID.\n\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_10m' for c in SPEC_COLS]  # Adding max columns\nFEATURES += [f'{c}_std_10m' for c in SPEC_COLS]  # Adding std columns\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_max_20s' for c in SPEC_COLS]  # Adding max columns\nFEATURES += [f'{c}_std_20s' for c in SPEC_COLS]  # Adding std columns\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ', end='')\n\n# A data matrix data is initialized to store the new features for each eeg_id in the train DataFrame.\n# For each row in train, the code calculates the mean, minimum, maximum, and standard deviation values within the specified 10-minute and 20-second windows.\n# These calculated values are then stored in the data matrix.\n# Finally, the matrix is added to the train DataFrame as new columns.\n\nif FEATURE_ENGINEER:\n    data = np.zeros((len(train), len(FEATURES)))\n    for k in range(len(train)):\n        if k % 100 == 0:\n            print(k, ', ', end='')\n        row = train.iloc[k]\n        r = int((row['min'] + row['max']) // 4)\n\n        # 10 MINUTE WINDOW FEATURES (MEANS, MINS, MAX, STD)\n        x = np.nanmean(spectrograms[row.spec_id][r:r + 300, :], axis=0)\n        data[k, :400] = x\n        x = np.nanmin(spectrograms[row.spec_id][r:r + 300, :], axis=0)\n        data[k, 400:800] = x\n        x = np.nanmax(spectrograms[row.spec_id][r:r + 300, :], axis=0)  # Calculate max\n        data[k, 800:1200] = x\n        x = np.nanstd(spectrograms[row.spec_id][r:r + 300, :], axis=0)  # Calculate std\n        data[k, 1200:1600] = x\n\n        # 20 SECOND WINDOW FEATURES (MEANS, MINS, MAX, STD)\n        x = np.nanmean(spectrograms[row.spec_id][r + 145:r + 155, :], axis=0)\n        data[k, 1600:2000] = x\n        x = np.nanmin(spectrograms[row.spec_id][r + 145:r + 155, :], axis=0)\n        data[k, 2000:2400] = x\n        x = np.nanmax(spectrograms[row.spec_id][r + 145:r + 155, :], axis=0)  # Calculate max\n        data[k, 2400:2800] = x\n        x = np.nanstd(spectrograms[row.spec_id][r + 145:r + 155, :], axis=0)  # Calculate std\n        data[k, 2800:3200] = x\n\n    train[FEATURES] = data\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\nprint()\nprint('New train shape:', train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:33:24.482691Z","iopub.execute_input":"2024-02-28T12:33:24.483344Z","iopub.status.idle":"2024-02-28T12:34:19.345276Z","shell.execute_reply.started":"2024-02-28T12:33:24.483304Z","shell.execute_reply":"2024-02-28T12:34:19.343880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:35:00.496427Z","iopub.execute_input":"2024-02-28T12:35:00.496953Z","iopub.status.idle":"2024-02-28T12:35:00.503123Z","shell.execute_reply.started":"2024-02-28T12:35:00.496893Z","shell.execute_reply":"2024-02-28T12:35:00.501881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_frequency_band_features(segment):\n    # Define EEG frequency bands\n    eeg_bands = {'Delta': (0.5, 4), 'Theta': (4, 8), 'Alpha': (8, 12), 'Beta': (12, 30), 'Gamma': (30, 45)}\n    \n    band_features = []\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        # Filter signal for the specific band\n        band_pass_filter = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        filtered = signal.sosfilt(band_pass_filter, segment)\n        # Extract features like mean, standard deviation, etc.\n        band_features.extend([np.nanmean(filtered), np.nanstd(filtered), np.nanmax(filtered), np.nanmin(filtered)])\n    \n    return band_features","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:35:14.089024Z","iopub.execute_input":"2024-02-28T12:35:14.089424Z","iopub.status.idle":"2024-02-28T12:35:14.099196Z","shell.execute_reply.started":"2024-02-28T12:35:14.089393Z","shell.execute_reply":"2024-02-28T12:35:14.097625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nimport time\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.decomposition import PCA  # Added import\n\nn_channels = sample_train_eeg.shape[1]\nn_channels\n\n# Initialize a PCA model\npca = PCA(n_components=0.95)\nprint(\"PCA model initialized.\")\n\n# Initialize an array for original features\nnum_rows = len(train)\nnum_features = 20 * n_channels  # 20 features per channel\ndata_original = np.zeros((num_rows, num_features))\n\nprint(\"Starting feature extraction and PCA processing...\")\nstart_time = time.time()\n\nfor k in range(num_rows):\n    if k % 1000 == 0:\n        print(f\"Processing row {k} of {num_rows}...\")\n    \n    row = train.iloc[k]\n    r = int((row['min'] + row['max']) // 4)\n    eeg_segment = spectrograms[row.spec_id][r:r+300, :]\n\n    # Apply the feature extraction function to each EEG channel\n    all_channel_features = []\n    for i in range(n_channels):\n        channel_features = extract_frequency_band_features(eeg_segment[:, i])\n        all_channel_features.extend(channel_features)\n\n    data_original[k, :] = all_channel_features\n\nprint(\"Data matrix constructed\")\n\n# Impute NaN values in the data matrix\nimputer = SimpleImputer(strategy='mean')\ndata_imputed = imputer.fit_transform(data_original)\n\nprint(f\"NaN values handled. Imputed data matrix shape: {data_imputed.shape}\")\n\n# Apply PCA on the imputed data\npca.fit(data_imputed)\nprint(\"PCA fitting completed.\")\n\n# Transform data using PCA\ndata_pca = pca.transform(data_imputed)\n\n# Add PCA features to DataFrame\npca_feature_columns = [f'pca_feature_{i}' for i in range(data_pca.shape[1])]\ntrain[pca_feature_columns] = data_pca\n\n# Measure total processing time\ntotal_time = time.time() - start_time\nprint(f\"Total processing time: {total_time:.2f} seconds.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-28T12:35:18.462223Z","iopub.execute_input":"2024-02-28T12:35:18.462727Z","iopub.status.idle":"2024-02-28T13:23:41.175300Z","shell.execute_reply.started":"2024-02-28T12:35:18.462693Z","shell.execute_reply":"2024-02-28T13:23:41.172551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T13:27:46.234319Z","iopub.execute_input":"2024-02-28T13:27:46.234903Z","iopub.status.idle":"2024-02-28T13:27:46.294753Z","shell.execute_reply.started":"2024-02-28T13:27:46.234861Z","shell.execute_reply":"2024-02-28T13:27:46.293588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Columns to be excluded from scaling\nexcluded_columns = ['eeg_id', 'spec_id', 'min', 'max', 'patient_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote','target']\n\n# Save the columns to be excluded\nexcluded_data = train[excluded_columns]\n\n# DataFrame with only the columns to be scaled\nfeatures = train.drop(columns=excluded_columns)\n\n# Initialize the StandardScaler\nscaler = StandardScaler()\n\n# Fit the scaler to the features and transform them\nfeatures_scaled = scaler.fit_transform(features)\n\n# Create a DataFrame from the scaled features\nfeatures_scaled_df = pd.DataFrame(features_scaled, columns=features.columns)\n\n# Concatenate the scaled features with the excluded columns\ntrain_scaled_df = pd.concat([excluded_data.reset_index(drop=True), features_scaled_df], axis=1)\ntrain_scaled_df\n","metadata":{"execution":{"iopub.status.busy":"2024-02-28T13:28:11.435177Z","iopub.execute_input":"2024-02-28T13:28:11.435626Z","iopub.status.idle":"2024-02-28T13:28:14.955640Z","shell.execute_reply.started":"2024-02-28T13:28:11.435594Z","shell.execute_reply":"2024-02-28T13:28:14.954751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_scaled_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T13:28:22.399026Z","iopub.execute_input":"2024-02-28T13:28:22.399502Z","iopub.status.idle":"2024-02-28T13:28:22.574369Z","shell.execute_reply.started":"2024-02-28T13:28:22.399463Z","shell.execute_reply":"2024-02-28T13:28:22.573007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\n\nprint('XGBoost version', xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T13:28:25.915019Z","iopub.execute_input":"2024-02-28T13:28:25.915429Z","iopub.status.idle":"2024-02-28T13:28:25.923378Z","shell.execute_reply.started":"2024-02-28T13:28:25.915399Z","shell.execute_reply":"2024-02-28T13:28:25.922018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VER = 1\nall_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train , train .target, train .patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = xgb.XGBClassifier(\n        objective='multi:softprob', \n        num_class=len(TARS),\n        learning_rate = 0.1, \n                      \n#         tree_method='gpu_hist',  #skip GPU acceleration\n    )\n    \n    # Prepare training and validation data\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              verbose=True, \n              early_stopping_rounds=10)\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)\n    \n   ","metadata":{"execution":{"iopub.status.busy":"2024-02-28T13:28:32.095297Z","iopub.execute_input":"2024-02-28T13:28:32.095761Z","iopub.status.idle":"2024-02-28T14:24:23.256111Z","shell.execute_reply.started":"2024-02-28T13:28:32.095725Z","shell.execute_reply":"2024-02-28T14:24:23.254961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOP = 30\n\n# Assuming 'model' is your trained model\nfeature_importance = model.feature_importances_\n\n# Get the feature names from 'train'\nfeature_names = train.columns\n\n# Sort the feature importances and get the indices of the sorted array\nsorted_idx = np.argsort(feature_importance)\n\n# Plot only the top 'TOP' features\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], feature_names[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T14:51:55.914629Z","iopub.execute_input":"2024-02-28T14:51:55.915165Z","iopub.status.idle":"2024-02-28T14:51:56.698704Z","shell.execute_reply.started":"2024-02-28T14:51:55.915130Z","shell.execute_reply":"2024-02-28T14:51:56.697300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T14:52:17.114663Z","iopub.execute_input":"2024-02-28T14:52:17.115110Z","iopub.status.idle":"2024-02-28T14:52:17.140494Z","shell.execute_reply.started":"2024-02-28T14:52:17.115075Z","shell.execute_reply":"2024-02-28T14:52:17.139216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n    x = np.nanmax( spec.iloc[:,1:].values, axis=0)\n    data[k, 800:1200] = x\n    x = np.nanstd( spec.iloc[:,1:].values, axis=0)\n    data[k,1200:1600] = x\n    \n    \n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1600:2000] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,2000:2400] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,2400:2800] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,2800:3200] = x\n\ntest[FEATURES] = data\nprint('New test shape',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T15:00:01.970097Z","iopub.execute_input":"2024-02-28T15:00:01.970679Z","iopub.status.idle":"2024-02-28T15:00:04.443238Z","shell.execute_reply.started":"2024-02-28T15:00:01.970635Z","shell.execute_reply":"2024-02-28T15:00:04.441964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER XGBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    # Load the XGBoost model\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    \n    # Make predictions\n    pred = model.predict_proba(test[FEATURES])\n    preds.append(pred)\n\n# Average the predictions from each fold\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T15:00:21.227281Z","iopub.execute_input":"2024-02-28T15:00:21.227752Z","iopub.status.idle":"2024-02-28T15:00:25.560097Z","shell.execute_reply.started":"2024-02-28T15:00:21.227714Z","shell.execute_reply":"2024-02-28T15:00:25.558902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T15:00:42.120325Z","iopub.execute_input":"2024-02-28T15:00:42.120760Z","iopub.status.idle":"2024-02-28T15:00:42.150152Z","shell.execute_reply.started":"2024-02-28T15:00:42.120728Z","shell.execute_reply":"2024-02-28T15:00:42.148948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.remove(\"/kaggle/working/XGB_v1_f2.model\")\nos.remove(\"/kaggle/working/XGB_v1_f0.model\")\nos.remove(\"/kaggle/working/XGB_v1_f1.model\")\nos.remove(\"/kaggle/working/XGB_v1_f4.model\")\nos.remove(\"/kaggle/working/XGB_v1_f3.model\")","metadata":{"execution":{"iopub.status.busy":"2024-02-28T15:16:13.113380Z","iopub.execute_input":"2024-02-28T15:16:13.114696Z","iopub.status.idle":"2024-02-28T15:16:13.125695Z","shell.execute_reply.started":"2024-02-28T15:16:13.114645Z","shell.execute_reply":"2024-02-28T15:16:13.124129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}