{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## HMS - Harmful Brain Activity Classification","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_dataset_directory = Path('/kaggle/input/hms-harmful-brain-activity-classification')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_categorical_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given categorical column in the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given categorical column\n\n    column: str\n        Name of the categorical column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    value_counts = df[column].value_counts()\n\n    fig, ax = plt.subplots(figsize=(24, df[column].value_counts().shape[0] + 4), dpi=100)\n    ax.bar(\n        x=np.arange(len(value_counts)),\n        height=value_counts.values,\n    )\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_xticks(\n        np.arange(len(value_counts)),\n        [\n            f'{value} ({count:,})' for value, count in value_counts.to_dict().items()\n        ]\n    )\n    ax.tick_params(axis='x', labelsize=15, pad=10)\n    ax.tick_params(axis='y', labelsize=15, pad=10)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n\n\ndef visualize_continuous_column_distribution(df, column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given continuous column in the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous column\n\n    column: str\n        Name of the continuous column,\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(24, 6), dpi=100)\n    ax.hist(df[column], bins=16)\n    ax.tick_params(axis='x', labelsize=15)\n    ax.tick_params(axis='y', labelsize=15)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.set_title(\n        title + f'''\n        Mean: {np.mean(df[column]):.2f} Median: {np.median(df[column]):.2f} Std: {np.std(df[column]):.2f}\n        Min: {np.min(df[column]):.2f} Max: {np.max(df[column]):.2f}\n        ''',\n        size=15,\n        pad=12.5,\n        loc='center',\n        wrap=True\n    )\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n        \ndef visualize_correlations(df, columns, title, path=None):\n\n    \"\"\"\n    Visualize correlations of given columns in the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given columns\n\n    columns: list\n        List of names of columns\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, ax = plt.subplots(figsize=(20, 20), dpi=100)\n    ax = sns.heatmap(\n        df[columns].corr(),\n        annot=True,\n        square=True,\n        cmap='coolwarm',\n        annot_kws={'size': 12},\n        fmt='.2f'\n    )\n    cbar = ax.collections[0].colorbar\n    cbar.ax.tick_params(labelsize=15)\n    ax.tick_params(axis='x', labelsize=10)\n    ax.tick_params(axis='y', labelsize=10)\n    ax.set_title(title, size=20, pad=15)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Introduction\n\nThe goal of this competition is to detect and classify seizures and other types of harmful brain activity in electroencephalography (EEG) data.\n\nDataset consist of EEG and spectrogram data. Each row on the training set represents:\n* **50 second** long EEG sample\n* Matched spectrogram of the EEG that covers a **10 minute** window\n\nBoth EEG and its matched spectrogram are centered at the same time. EEG files are more than spectrogram files because many of the raw samples were overlapping and some of them were merged into single files. Metadata on training set allows you to extract the original subsets using \n`eeg_label_offset_seconds` and `spectrogram_label_offset_seconds` columns.","metadata":{}},{"cell_type":"code","source":"eeg_directory = competition_dataset_directory / 'train_eegs'\nspectrogram_directory = competition_dataset_directory / 'train_spectrograms'\ndf_train = pd.read_csv(competition_dataset_directory / 'train.csv')\n\nprint(f'Training Set Shape: {df_train.shape} - EEG Files: {len(os.listdir(eeg_directory))} - Spectrogram Files: {len(os.listdir(spectrogram_directory))}')\n\ndf_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Targets\n\nTarget columns in training set are:\n\n* `seizure_vote`: Seizure\n* `lpd_vote`: Lateralized periodic discharges\n* `gpd_vote`: Generalized periodic discharges\n* `lrda_vote`: Lateralized rhythmic delta activity\n* `grda_vote`: Generalized rhythmic delta activity\n* `other_vote`: Other\n\nThose target columns represent counts of annotator votes for a given brain activity class. There is also another column named `expert_consensus` which is the argmax of previously mentioned target columns. It is provided for convenience only. Value counts of `expert_consensus` are balanced to some extend.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train,\n    column='expert_consensus',\n    title='expert_consensus Counts'    \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Number of annotators varies between 1 and 28 with an average of **7.26**. Since the number of annotators is not consistent in training samples, target columns should be evaluated based on vote ratios rather than counts.","metadata":{}},{"cell_type":"code","source":"target_columns = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ndf_train['total_vote'] = df_train[target_columns].sum(axis=1)\n\nvisualize_continuous_column_distribution(\n    df=df_train,\n    column='total_vote',\n    title='total_vote Histogram'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Histograms of normalized target columns are visualized below. Target columns with high 0 and 0.7-1 bins indicate that annotators were having less disagreements while labeling those brain activities. On the other hand, target columns with high 0.1-0.6 bins indicate that brain activity is hard to detect and annotators had more disagreements.","metadata":{}},{"cell_type":"code","source":"normalized_target_columns = [f'{column}_normalized' for column in target_columns]\ndf_train[normalized_target_columns] = df_train[target_columns] / df_train['total_vote'].values.reshape(-1, 1)\n\nfor column in normalized_target_columns:\n    visualize_continuous_column_distribution(\n        df=df_train,\n        column=column,\n        title=f'Normalized Target Column {column}'\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Roughly half of the training set votes are unanimous. The count of samples with unamious vote is **51037** and the count of samples with vote disagreements is **55763**.","metadata":{}},{"cell_type":"code","source":"df_train['unanimous_vote'] = (df_train[normalized_target_columns] == 1.0).any(axis=1).astype(np.uint8)\n\nvisualize_categorical_column_distribution(\n    df=df_train,\n    column='unanimous_vote',\n    title='unanimous_vote Counts'    \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Annotator disagreements are very common in medical domain because tasks are challenging even for the experts. However there could be underlying patterns in their annotations. For example, targets don't have similar number of unanimous votes which clearly shows that experts had more disagreements on certain brain activities such as lateralized and generalized periodic discharges. On the contrary, experts had less disagreements while they were detecting seizures.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_train.loc[df_train['unanimous_vote'] == 1],\n    column='expert_consensus',\n    title='Unanimous expert_consensus Counts'    \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Correlations of normalized target columns can be seen below. Target columns like `Seizure`, `GRDA` and `Other` have more blue intensity, thus their vote percentages are higher.","metadata":{}},{"cell_type":"code","source":"visualize_correlations(\n    df=df_train,\n    columns=normalized_target_columns,\n    title='Normalized Target Columns Correlations',\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. EEGs\n\nEEG stands for electroencephalogram. It is a test that measures and records the electrical activity in the brain. The brain cells communicate with each other through electrical impulses, and these electrical signals can be detected and recorded using electrodes placed on the scalp.\n\nEEG is commonly used in clinical settings to diagnose and monitor various neurological disorders. In this case, the neurological disorders are seizure, generalized/lateralized periodic discharges, lateralized/generalized rhythmic delta activity and other.\n\nThere are **1950** patients in training set. Each patient has **8.76** EEGs on average and each EEG has **6.25** subsamples on average which adds up to **106800** labeled 50 second long EEG subsamples.","metadata":{}},{"cell_type":"code","source":"df_patient_id_eeg_id_unique_counts = df_train.groupby('patient_id')[['eeg_id']].nunique()\nvisualize_continuous_column_distribution(\n    df=df_patient_id_eeg_id_unique_counts,\n    column='eeg_id',\n    title='patient_id eeg_id nunique Distribution'\n)\n\ndf_eeg_id_eeg_sub_id_unique_counts = df_train.groupby('eeg_id')[['eeg_sub_id']].nunique()\nvisualize_continuous_column_distribution(\n    df=df_eeg_id_eeg_sub_id_unique_counts,\n    column='eeg_sub_id',\n    title='eeg_id eeg_sub_id nunique Distribution'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EEGs in this dataset use 10-20 system. The 10-20 system is a standardized method for electrode placement. It is widely used to ensure consistency and accuracy in EEG recordings across different laboratories and healthcare settings. The name \"10-20\" refers to the distances between adjacent electrode placements, which are either 10% or 20% of the total front-back or right-left distance of the skull.\n\nThe 10-20 system is widely used in clinical and research settings because it provides a standardized and reproducible way to place electrodes on the scalp. This consistency is essential for comparing EEG recordings across different individuals and studies.\n\nOther variations of the system, such as the 10-10 system, involve additional electrode placements for more detailed mapping of the scalp. The choice of system depends on the specific requirements of the EEG recording and the clinical or research goals.\n\nElectrode placements of 10-20 system can be seen on the image below.\n\n![1020system](https://i.ibb.co/KrL8mGp/Screenshot-from-2024-01-13-13-34-15.png)\n\nHowever, signals of A1 and A2 electrodes are not included but instead an additional EKG signal is added in this dataset. The EKG is for an electrocardiogram lead that records data from the heart. All of the columns in EEGs represent electrical signal on corresponding electrodes and `EKG` column represents the electrical signal on heart over time.","metadata":{}},{"cell_type":"code","source":"df_eeg = pd.read_parquet(eeg_directory / '1628180742.parquet')\ndf_eeg","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All of the EEG data (for both train and test) was collected at a frequency of 200 samples per second so EEG `1628180742` is 90 seconds long. A `second` identifier column can be created by repeating each second 200 times.","metadata":{}},{"cell_type":"code","source":"sample_per_second = 200\nseconds = df_eeg.shape[0] / sample_per_second\n\ndf_eeg['second'] = np.repeat(np.arange(seconds), 200)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The entire EEG signal of `1628180742` is visualized below.","metadata":{}},{"cell_type":"code","source":"def visualize_eeg_signal(df, title, path=None):\n    \n    \"\"\"\n    Visualize EEG signal over time\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given EEG and EKG columns\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n    \n    eeg_columns = [\n        'Fp1', 'F3', 'C3', 'P3', 'F7', 'T3',\n        'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2',\n        'F4', 'C4', 'P4', 'F8', 'T4', 'T6',\n        'O2',\n    ]\n    ekg_column = 'EKG'\n    eeg_spacing = 500\n    \n    fig, axes = plt.subplots(figsize=(24, 24), nrows=2, height_ratios=[10, 1], dpi=100)\n\n    for column_idx, column in enumerate(eeg_columns):\n        axes[0].plot(np.arange(0, df.shape[0]), df[column] + (eeg_spacing * column_idx), linewidth=0.5, color='black')\n        \n    y_ticks = np.arange(0, len(eeg_columns)) * eeg_spacing - 100\n    axes[0].set_yticks(y_ticks)\n    axes[0].set_yticklabels(eeg_columns)\n    axes[0].tick_params(axis='x', labelsize=15)\n    axes[0].tick_params(axis='y', labelsize=15)\n    axes[0].set_xlabel('')\n    axes[0].set_ylabel('')\n    axes[0].set_title(title, size=15, pad=12.5, loc='center')\n    \n    axes[1].plot(np.arange(0, df.shape[0]), df['EKG'], linewidth=0.5, color='black')\n    axes[1].set_yticks(np.array(axes[1].get_yticks()) * 1.5)\n    axes[1].tick_params(axis='x', labelsize=12.5)\n    axes[1].tick_params(axis='y', labelsize=12.5)\n    axes[1].set_xlabel('')\n    axes[1].set_ylabel('')\n    axes[1].set_title('EKG', size=15, pad=12.5, loc='center')\n    \n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n\nvisualize_eeg_signal(\n    df=df_eeg,\n    title='EEG 1628180742'\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:22:19.531168Z","iopub.execute_input":"2024-02-05T08:22:19.531669Z","iopub.status.idle":"2024-02-05T08:22:20.934639Z","shell.execute_reply.started":"2024-02-05T08:22:19.531629Z","shell.execute_reply":"2024-02-05T08:22:20.933639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 9 subsamples exist for EEG `1628180742`. They can be retrieved using the metadata from training set. `eeg_label_offset_seconds` is the start of that subsample and each subsample is 50 seconds long. Since the **central 10 seconds** are labeled for both EEGs and spectrograms, other parts may have overlap with other subsamples.","metadata":{}},{"cell_type":"code","source":"df_train.loc[df_train['eeg_id'] == 1628180742]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All subsamples of EEG `1628180742` are visualized below.","metadata":{}},{"cell_type":"code","source":"for idx, row in df_train.loc[df_train['eeg_id'] == 1628180742].reset_index(drop=True).iterrows():\n    \n    start_idx = int(row['eeg_label_offset_seconds'] * 200)\n    end_idx = int((row['eeg_label_offset_seconds'] + 50) * 200)\n    \n    visualize_eeg_signal(\n        df=df_eeg.iloc[start_idx:end_idx],\n        title=f'EEG {row[\"eeg_id\"]} - Subsample {row[\"eeg_sub_id\"]} ({start_idx}-{end_idx}) - Label {row[\"expert_consensus\"]}'\n    )\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Overlapping EEG subsamples might be problematic if they have inconsistent annotations. They can be found by checking the difference of `eeg_label_offset_seconds` and `expert_consensus`.","metadata":{}},{"cell_type":"code","source":"df_train['expert_consensus_encoded'] = df_train['expert_consensus'].map({\n    'Seizure': 1,\n    'LPD': 2,\n    'GPD': 3,\n    'LRDA': 4,\n    'GRDA': 5,\n    'Other': 6,\n})\n\ndf_train['eeg_label_offset_seconds_diff'] = df_train.groupby('eeg_id')['eeg_label_offset_seconds'].diff()\ndf_train['expert_consensus_encoded_diff'] = df_train.groupby('eeg_id')['expert_consensus_encoded'].diff()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 1551 subsamples that have different labels when they have overlapping seconds between 2 and 48. `eeg_label_offset_seconds_diff == x` means that row has an overlapping `50 - x` amount of seconds with another row and their targets are different. Since the subsample labels are based on the center 10 seconds, overlapping seconds between 2 and 40 don't cause any inconsistent labels. Target counts of 50 second overlapping subsamples are consistent with the entire dataset.","metadata":{}},{"cell_type":"code","source":"condition = (df_train['eeg_label_offset_seconds_diff'] < 50) & (df_train['expert_consensus_encoded_diff'] != 0)\n\nvisualize_categorical_column_distribution(\n    df=df_train.loc[condition],\n    column='expert_consensus',\n    title='Overlapping 50 Second Subsamples expert_consensus Counts'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Label inconsistency arises when subsamples have more than 40 seconds overlap because their center 10 seconds are starting to overlap after that point. There are **594** subsamples that have more than **40** seconds overlap with different target values. **109** of those subsamples even have 48 seconds of overlap with different target values. Those inconsistencies should be dealth with accordingly.","metadata":{}},{"cell_type":"code","source":"condition = (df_train['eeg_label_offset_seconds_diff'] < 10) & (df_train['expert_consensus_encoded_diff'] != 0)\n\nvisualize_categorical_column_distribution(\n    df=df_train.loc[condition],\n    column='eeg_label_offset_seconds_diff',\n    title='Overlapping 10 Second Subsamples eeg_label_offset_seconds_diff Counts'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. EEG Metadata\n\nData that is coming from multiple sources is very common in medical domain and that could be the case for EEGs. Patients, EEG machine, configurations and etc. can lead to inconsistencies in data collection process. The code below is used for metadata extraction in order to compare EEG subsamples.  ","metadata":{}},{"cell_type":"code","source":"df_eeg_metadata = []\n\nfor eeg_id, df_train_eeg in tqdm(df_train.groupby('eeg_id'), total=df_train['eeg_id'].nunique()):\n\n    df_eeg = pd.read_parquet(eeg_directory / f'{eeg_id}.parquet')\n\n    for _, row in df_train_eeg.iterrows():\n\n        eeg_sub_id = row['eeg_sub_id']\n        start_idx = int(row['eeg_label_offset_seconds'] * 200)\n        end_idx = int((row['eeg_label_offset_seconds'] + 50) * 200)\n        df_eeg_subsample = df_eeg.iloc[start_idx:end_idx].reset_index(drop=True)\n\n        nan_counts = df_eeg_subsample.isnull().sum().to_dict()\n        nan_counts = {f'{k}_nan_count': v for k, v in nan_counts.items()}\n        means = df_eeg_subsample.mean(axis=0).to_dict()\n        means = {f'{k}_mean': v for k, v in means.items()}\n        stds = df_eeg_subsample.std(axis=0).to_dict()\n        stds = {f'{k}_std': v for k, v in stds.items()}\n\n        metadata_dict = {\n            'eeg_id': eeg_id,\n            'eeg_sub_id': eeg_sub_id,\n            'row_count': df_eeg_subsample.shape[0],\n            'column_count': df_eeg_subsample.shape[1]\n        }\n        metadata_dict.update(nan_counts)\n        metadata_dict.update(means)\n        metadata_dict.update(stds)\n        df_eeg_metadata.append(metadata_dict)\n\ndf_eeg_metadata = pd.DataFrame(df_eeg_metadata)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T07:13:19.116672Z","iopub.execute_input":"2024-02-05T07:13:19.118066Z","iopub.status.idle":"2024-02-05T07:28:54.759022Z","shell.execute_reply.started":"2024-02-05T07:13:19.118016Z","shell.execute_reply":"2024-02-05T07:28:54.757954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All of the EEG subsamples have consistent number of channels and timesteps.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_eeg_metadata,\n    column='row_count',\n    title='EEG Metadata row_count Counts'    \n)\n\nvisualize_categorical_column_distribution(\n    df=df_eeg_metadata,\n    column='column_count',\n    title='EEG Metadata column_count Counts'    \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mean and standard deviation histograms of EEG subsamples suggest that EEGs are on different scales. Instead of normalizing with dataset statistics, instance normalization makes more sense.\n\nMissing count histograms of EEG subsamples are identical on all channels, so missing values either exist on all channels at the same time or they don't exist at all. Missing values could be related to EEG failures or it could be some kind of an anomaly.","metadata":{}},{"cell_type":"code","source":"def visualize_eeg_metadata_distribution(df, eeg_column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given continuous columns in the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous column\n\n    eeg_column: str\n        Name of the EEG column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, axes = plt.subplots(figsize=(24, 6), ncols=3, dpi=100)\n    axes[0].hist(df[f'{eeg_column}_mean'], bins=16)\n    axes[1].hist(df[f'{eeg_column}_std'], bins=16)\n    axes[2].hist(df[f'{eeg_column}_nan_count'], bins=16)\n    \n    for i, metadata in enumerate(['mean', 'std', 'nan_count']):\n        axes[i].tick_params(axis='x', labelsize=15)\n        axes[i].tick_params(axis='y', labelsize=15)\n        axes[i].set_xlabel('')\n        axes[i].set_ylabel('')\n        axes[i].set_title(\n            f'''\n            {metadata}\n            Mean: {np.mean(df[f'{eeg_column}_{metadata}']):.2f} Median: {np.median(df[f'{eeg_column}_{metadata}']):.2f} Std: {np.std(df[f'{eeg_column}_{metadata}']):.2f}\n            Min: {np.min(df[f'{eeg_column}_{metadata}']):.2f} Max: {np.max(df[f'{eeg_column}_{metadata}']):.2f}\n            ''',\n            size=15,\n            pad=12.5,\n            loc='center',\n            wrap=True\n        )\n        \n    fig.suptitle(title, fontsize=20, y=1.2)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n\neeg_ekg_columns = [\n    'Fp1', 'F3', 'C3', 'P3', 'F7', 'T3',\n    'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2',\n    'F4', 'C4', 'P4', 'F8', 'T4', 'T6',\n    'O2', 'EKG'\n]\n\nfor column in eeg_ekg_columns:\n    visualize_eeg_metadata_distribution(\n        df=df_eeg_metadata,\n        eeg_column=column,\n        title=f'EEG {column} Metadata'\n    )\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Spectrograms\n\nA spectrogram is a detailed view of an EEG, is able to represent time, frequency, and amplitude at the same time. It is a way to analyze the frequency content of a signal over time. Spectrograms are commonly used in signal processing, audio analysis, and other fields where understanding the frequency distribution of a signal is important.\n\nA raw EEG is a waveform that displays changes in a signal’s amplitude over time. A spectrogram, however, displays changes in the frequencies in a signal over time. Amplitude is then represented as variable intensity.\n\nThere are **1950** patients in training set. Each patient has **5.71** spectrograms on average and each spectrogram has **9.59** subsamples on average which adds up to 106800 labeled 10 minute long spectrogram subsamples.","metadata":{}},{"cell_type":"code","source":"df_patient_id_spectrogram_id_unique_counts = df_train.groupby('patient_id')[['spectrogram_id']].nunique()\nvisualize_continuous_column_distribution(\n    df=df_patient_id_spectrogram_id_unique_counts,\n    column='spectrogram_id',\n    title='patient_id spectrogram_id nunique Distribution'\n)\n\ndf_spectromram_id_spectrogram_sub_id_unique_counts = df_train.groupby('spectrogram_id')[['spectrogram_sub_id']].nunique()\nvisualize_continuous_column_distribution(\n    df=df_spectromram_id_spectrogram_sub_id_unique_counts,\n    column='spectrogram_sub_id',\n    title='spectrogram_id spectrogram_sub_id nunique Distribution'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectrograms in this dataset are montages of **10 minute** long EEGs, not the **50 second** long ones that are given. Thus, they are zoomed out versions of given raw EEGs.\n\nMontages are arrangements of signals that are created to display activity over the entire head and to provide lateralizing and localizing information. There are two categories of montages which are bipolar and referential. There are also multiple types of bipolar montages, but the most common one is the double banana, in which each electrode is linked and compared to the one behind it.\n\nIn the double banana montage, there are two chains per side\n\n* Left outside temporal chain involving Fp1 → F7 → T3 → T5 → O1\n* Left inside parasagittal chain involving Fp1 → F3 → C3 → P3 → O1\n* Right outside temporal chain involving Fp2 → F8 → T4 → T6 → O2\n* Right inside parasagittal chain involving Fp2 → F4 → C4 → P4 → O2\n\nFinally, the \"z\" electrodes Fz → Cz → Pz form a small central chain.","metadata":{}},{"cell_type":"code","source":"df_spectrogram = pd.read_parquet(spectrogram_directory / '353733.parquet')\ndf_spectrogram","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectrogram files have **400 signal** and **1 time** columns in total.\n\nSignal column names start with `LL` (left temporal), `LP` (left parasagittal), `RL` (right temporal) and `RP` (right parasagittal) correspond to chains that are listed above, Small central chain and the EKG signal are not included. \n\nSignal column names end with a floating point number that represents the frequency in hertz (Hz). Those numbers are between 0.59 and 19.92 with a step size of 0.19/0.20.\n\nFinally, there is a column named `time` which represents the time in **2** seconds, so a spectrogram with 300 rows is 600 seconds (10 minutes) long.","metadata":{}},{"cell_type":"code","source":"signal_column_frequencies = [float((a).split('_')[1]) for a in df_spectrogram.columns.tolist()[1:101]]\nprint(signal_column_frequencies)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:33:46.968078Z","iopub.execute_input":"2024-02-05T08:33:46.969986Z","iopub.status.idle":"2024-02-05T08:33:46.979292Z","shell.execute_reply.started":"2024-02-05T08:33:46.969925Z","shell.execute_reply":"2024-02-05T08:33:46.977412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectrogram subsamples can be retrieved by taking 600 seconds after the start time (`spectrogram_label_offset_seconds` column).\n\nThe function below can be used to get spectrogram subsamples by passing the spectrogram dataframe and start time. First column (`time`) is not taken and data is transposed. Returned spectrogram array has frequency on the first axis and time on the second axis. Different chains are stacked on the channel axis.","metadata":{}},{"cell_type":"code","source":"def get_spectrogram(df_spectrogram, start_time, fill_na=False, log_scale=False):\n    \n    \"\"\"\n    Get spectrogram array from the dataframe\n\n    Parameters\n    ----------\n    df_spectrogram: pandas.DataFrame\n        Dataframe with time and signal columns\n\n    start_time: int\n        Spectrogram offset seconds\n        \n    fill_na: bool\n        Whether to fill missing values or not\n\n    log_scale: bool\n        Whether to do log transform or not\n\n    Returns\n    -------\n    spectrogram: numpy.ndarray of shape (n_frequencies, n_time_steps, n_channels)\n        Array of spectrogram\n    \"\"\"\n    \n    if start_time % 2 == 0:\n        start_time += 1\n    end_time = start_time + 598\n    \n    df_spectrogram_subsample = df_spectrogram.loc[(df_spectrogram['time'] >= start_time) & (df_spectrogram['time'] <= end_time)].iloc[:, 1:]\n    \n    if fill_na:\n        df_spectrogram_subsample = df_spectrogram_subsample.fillna(0)\n    \n    spectrogram = df_spectrogram_subsample.values.T\n    spectrogram = np.stack((\n        spectrogram[0:100, :],\n        spectrogram[100:200, :],\n        spectrogram[200:300, :],\n        spectrogram[300:400, :],\n    ), axis=-1)\n    \n    if log_scale:\n        spectrogram = np.log1p(spectrogram)\n        \n    return spectrogram\n\n\nspectrogram = get_spectrogram(\n    df_spectrogram=df_spectrogram,\n    start_time=0,\n    fill_na=False,\n    log_scale=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:33:44.114362Z","iopub.execute_input":"2024-02-05T08:33:44.114867Z","iopub.status.idle":"2024-02-05T08:33:44.129710Z","shell.execute_reply.started":"2024-02-05T08:33:44.114828Z","shell.execute_reply":"2024-02-05T08:33:44.127968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A spectrogram with seizure label is visualized below. Since the center 10 seconds are labeled, high amplitude in the center of right chains can be seen clearly. However, similar kind of high amplitude can't be seen on the left chains, so it means that activations can be anywhere including single electrodes.","metadata":{}},{"cell_type":"code","source":"def visualize_spectrogram(spectrogram, frequencies, channel_names, path=None):\n    \n    \"\"\"\n    Visualize spectrogram\n\n    Parameters\n    ----------\n    spectrogram: numpy.ndarray of shape (n_frequencies, n_time_steps, n_channels)\n        Array of spectrogram\n\n    frequencies: list of shape (n_frequencies)\n        List of frequencies\n        \n    channel_names: list of shape (n_channels)\n        List of channel names\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n    \n    channels = spectrogram.shape[2]\n    fig, axes = plt.subplots(figsize=(32, 32), nrows=channels)\n    for channel in range(channels):\n        spectrogram_channel = spectrogram[:, :, channel]\n        axes[channel].imshow(spectrogram_channel, cmap='inferno')\n        axes[channel].set_yticks(np.arange(0, spectrogram.shape[0])[::5])\n        axes[channel].set_yticklabels(frequencies[::5])\n        axes[channel].tick_params(axis='x', labelsize=15)\n        axes[channel].tick_params(axis='y', labelsize=15)\n        axes[channel].set_xlabel('Time', size=15, labelpad=12.5)\n        axes[channel].set_ylabel('Frequency (Hz)', size=15, labelpad=12.5)\n        axes[channel].set_title(f'{channel_names[channel]} Spectrogram', size=15, pad=12.5, loc='center')\n        \n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n        \n\nchannel_names = [\n    'LL (left temporal chain)',\n    'RL (right temporal chain)',\n    'LP (left parasagittal chain)',\n    'RP (right parasagittal chain)'\n]\n\nvisualize_spectrogram(\n    spectrogram,\n    frequencies=signal_column_frequencies,\n    channel_names=channel_names\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:22:29.483485Z","iopub.execute_input":"2024-02-05T08:22:29.483934Z","iopub.status.idle":"2024-02-05T08:22:29.541879Z","shell.execute_reply.started":"2024-02-05T08:22:29.483901Z","shell.execute_reply":"2024-02-05T08:22:29.540017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Spectrogram Metadata\n\nThe code below is used for metadata extraction in order to compare spectrogram subsamples.","metadata":{}},{"cell_type":"code","source":"df_spectrogram_metadata = []\n\nfor spectrogram_id, df_train_spectrogram in tqdm(df_train.groupby('spectrogram_id'), total=df_train['spectrogram_id'].nunique()):\n\n    df_spectrogram = pd.read_parquet(spectrogram_directory / f'{spectrogram_id}.parquet')\n\n    for _, row in df_train_spectrogram.iterrows():\n\n        spectrogram_sub_id = row['spectrogram_sub_id']\n        start_time = row['spectrogram_label_offset_seconds']\n        \n        if start_time % 2 == 0:\n            start_time += 1\n        end_time = start_time + 598\n\n        df_spectrogram_subsample = df_spectrogram.loc[(df_spectrogram['time'] >= start_time) & (df_spectrogram['time'] <= end_time)].iloc[:, 1:]\n\n        nan_counts = df_spectrogram_subsample.isnull().sum().to_dict()\n        nan_counts = {f'{k}_nan_count': v for k, v in nan_counts.items()}\n        means = df_spectrogram_subsample.mean(axis=0).to_dict()\n        means = {f'{k}_mean': v for k, v in means.items()}\n        stds = df_spectrogram_subsample.std(axis=0).to_dict()\n        stds = {f'{k}_std': v for k, v in stds.items()}\n\n        metadata_dict = {\n            'spectrogram_id': spectrogram_id,\n            'spectrogram_sub_id': spectrogram_sub_id,\n            'row_count': df_spectrogram_subsample.shape[0],\n            'column_count': df_spectrogram_subsample.shape[1]\n        }\n        metadata_dict.update(nan_counts)\n        metadata_dict.update(means)\n        metadata_dict.update(stds)\n        df_spectrogram_metadata.append(metadata_dict)\n\ndf_spectrogram_metadata = pd.DataFrame(df_spectrogram_metadata)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T07:28:54.761266Z","iopub.execute_input":"2024-02-05T07:28:54.762390Z","iopub.status.idle":"2024-02-05T07:52:04.500381Z","shell.execute_reply.started":"2024-02-05T07:28:54.762352Z","shell.execute_reply":"2024-02-05T07:52:04.498872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All of the spectrogram subsamples have consistent number of frequencies and timesteps.","metadata":{}},{"cell_type":"code","source":"visualize_categorical_column_distribution(\n    df=df_spectrogram_metadata,\n    column='row_count',\n    title='Spectrogram Metadata row_count Counts'    \n)\n\nvisualize_categorical_column_distribution(\n    df=df_spectrogram_metadata,\n    column='column_count',\n    title='Spectrogram Metadata column_count Counts'    \n)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:05:27.762771Z","iopub.execute_input":"2024-02-05T08:05:27.763250Z","iopub.status.idle":"2024-02-05T08:05:28.374662Z","shell.execute_reply.started":"2024-02-05T08:05:27.763218Z","shell.execute_reply":"2024-02-05T08:05:28.373074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mean and standard deviation histograms of spectrogram subsamples show that frequencies are on different scales. Lower frequencies have higher amplitudes, so they also have higher mean and standard deviations. As frequencies increase, their statistics decrease consistently but there are some minor exceptions.\n\nMissing count histograms of spectrogram subsamples are same on all chains and frequencies, so missing values always occur at the same time or they don't exist at all just like EEGs.","metadata":{}},{"cell_type":"code","source":"def visualize_spectrogram_metadata_distribution(df, spectrogram_column, title, path=None):\n\n    \"\"\"\n    Visualize distribution of the given continuous columns in the given dataframe\n\n    Parameters\n    ----------\n    df: pandas.DataFrame\n        Dataframe with given continuous column\n\n    spectrogram_column: str\n        Name of the spectrogram column\n\n    title: str\n        Title of the plot\n\n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    fig, axes = plt.subplots(figsize=(24, 6), ncols=3, dpi=100)\n    axes[0].hist(df[f'{spectrogram_column}_mean'], bins=16)\n    axes[1].hist(df[f'{spectrogram_column}_std'], bins=16)\n    axes[2].hist(df[f'{spectrogram_column}_nan_count'], bins=16)\n    \n    for i, metadata in enumerate(['mean', 'std', 'nan_count']):\n        axes[i].tick_params(axis='x', labelsize=15)\n        axes[i].tick_params(axis='y', labelsize=15)\n        axes[i].set_xlabel('')\n        axes[i].set_ylabel('')\n        axes[i].set_title(\n            f'''\n            {metadata}\n            Mean: {np.mean(df[f'{spectrogram_column}_{metadata}']):.2f} Median: {np.median(df[f'{spectrogram_column}_{metadata}']):.2f} Std: {np.std(df[f'{spectrogram_column}_{metadata}']):.2f}\n            Min: {np.min(df[f'{spectrogram_column}_{metadata}']):.2f} Max: {np.max(df[f'{spectrogram_column}_{metadata}']):.2f}\n            ''',\n            size=15,\n            pad=12.5,\n            loc='center',\n            wrap=True\n        )\n        \n    fig.suptitle(title, fontsize=20, y=1.2)\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path, bbox_inches='tight')\n        plt.close(fig)\n\n\nspectrogram_columns = [[f'{chain}_{frequency}' for frequency in signal_column_frequencies] for chain in ['LL', 'RL', 'LP', 'RP']]\nspectrogram_columns = np.array(spectrogram_columns).flatten().tolist()\n\nfor column in spectrogram_columns:\n    visualize_spectrogram_metadata_distribution(\n        df=df_spectrogram_metadata,\n        spectrogram_column=column,\n        title=f'Spectrogram {column} Metadata'\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Missing Values\n\nMissing values exist in a small subset of both EEG and spectrogram subsamples. They always occur on the same EEG channels or spectrogram chains and frequencies, so they are consistent on time steps.","metadata":{}},{"cell_type":"code","source":"eeg_nan_count_columns = [column for column in df_eeg_metadata.columns.tolist() if 'nan_count' in column]\nspectrogram_nan_count_columns = [column for column in df_spectrogram_metadata.columns.tolist() if 'nan_count' in column]\n\ndf_na_eegs = df_eeg_metadata.loc[(df_eeg_metadata[eeg_nan_count_columns] > 0).any(axis=1), ['eeg_id', 'eeg_sub_id'] + eeg_nan_count_columns]\ndf_na_eegs = df_na_eegs.sort_values(by='Fp1_nan_count', ascending=False).reset_index(drop=True)\n\ndf_na_spectrograms = df_spectrogram_metadata.loc[(df_spectrogram_metadata[spectrogram_nan_count_columns] > 0).any(axis=1), ['spectrogram_id', 'spectrogram_sub_id'] + spectrogram_nan_count_columns]\ndf_na_spectrograms = df_na_spectrograms.sort_values(by='LL_0.59_nan_count', ascending=False).reset_index(drop=True)\n\nprint(f'There are {df_na_eegs.shape[0]} EEG and {df_na_spectrograms.shape[0]} spectrogram subsamples with at least 1 missing value')\nprint(f'Subsamples are from {df_na_eegs[\"eeg_id\"].nunique()} EEGs and {df_na_spectrograms[\"spectrogram_id\"].nunique()} spectrograms')\nprint(f'{(df_na_eegs[\"eeg_id\"].nunique() / df_train[\"eeg_id\"].nunique() * 100):.2f}% of EEG and {(df_na_spectrograms[\"spectrogram_id\"].nunique() / df_train[\"spectrogram_id\"].nunique() * 100):.2f}% of spectrogram subsamples have at least 1 missing value')","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:11:11.169158Z","iopub.execute_input":"2024-02-05T08:11:11.171195Z","iopub.status.idle":"2024-02-05T08:11:12.006205Z","shell.execute_reply.started":"2024-02-05T08:11:11.171130Z","shell.execute_reply":"2024-02-05T08:11:12.004860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Number of missing values statistics are calculated on subsamples with at least 1 missing value. Average missing value count is **170.75** on EEGs and **75.63** on spectrograms. Maximum number of missing values is **5197** for EEGs and **296** for spectrograms.\n\nSince there are less time steps on spectrograms, they have more missing values compared to EEGs. Besides, number of spectrogram subsamples with at least 1 missing value is greater than 2 times of number of EEG subsamples with at least 1 missing value.\n\nThe reason of this phenomenon is probably related to spectrogram subsamples cover 10 minutes while EEG subsamples cover 50 seconds. It is more likely to happen in larger time frames.","metadata":{}},{"cell_type":"code","source":"visualize_continuous_column_distribution(\n    df=df_na_eegs,\n    column='Fp1_nan_count',\n    title='EEG Subsamples NA Counts',\n)\n\nvisualize_continuous_column_distribution(\n    df=df_na_spectrograms,\n    column='LL_0.59_nan_count',\n    title='Spectrogram Subsamples NA Counts',\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:18:54.704090Z","iopub.execute_input":"2024-02-05T08:18:54.704723Z","iopub.status.idle":"2024-02-05T08:18:55.992120Z","shell.execute_reply.started":"2024-02-05T08:18:54.704677Z","shell.execute_reply":"2024-02-05T08:18:55.990169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EEG subsamples with top 5 most missing values are visualized below. All of the missing values exist either on the most left or right side on all of them. They aren't randomly scattered around.\n\nEEG 1593385762 also shows that there are different kind of anomalies that should be handled properly.","metadata":{}},{"cell_type":"code","source":"for eeg_id, eeg_label_offset_seconds in [(3289898692, 36.0), (3931449367, 0), (1593385762, 0), (2190373347, 0), (975631111, 0)]:\n    \n    start_idx = int(eeg_label_offset_seconds * 200)\n    end_idx = int((eeg_label_offset_seconds + 50) * 200)\n    \n    df_eeg = pd.read_parquet(eeg_directory / f'{eeg_id}.parquet')\n    \n    visualize_eeg_signal(\n        df=df_eeg.iloc[start_idx:end_idx],\n        title=f'EEG {eeg_id} - {df_na_eegs.loc[df_na_eegs[\"eeg_id\"] == eeg_id, \"Fp1_nan_count\"].values[0]} Missing Values'\n    )\n","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:23:37.295833Z","iopub.execute_input":"2024-02-05T08:23:37.296401Z","iopub.status.idle":"2024-02-05T08:23:43.717155Z","shell.execute_reply.started":"2024-02-05T08:23:37.296352Z","shell.execute_reply":"2024-02-05T08:23:43.714532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectrogram subsamples with top 5 most missing values are visualized below. Unlike EEGs, all of the missing values exist on the both sides of the center 10 seconds which is the labeled part. It could be an indication of the EEG device being inactive, thus no amplitude was present.","metadata":{}},{"cell_type":"code","source":"for spectrogram_id, spectrogram_label_offset_seconds in [(1925646794, 0), (314642970, 0), (1152072732, 0), (1250083997, 0), (1190299410, 0)]:\n        \n    df_spectrogram = pd.read_parquet(spectrogram_directory / f'{spectrogram_id}.parquet')\n    spectrogram = get_spectrogram(\n        df_spectrogram=df_spectrogram,\n        start_time=spectrogram_label_offset_seconds,\n        fill_na=False,\n        log_scale=False\n    )\n    \n    visualize_spectrogram(\n        spectrogram=spectrogram,\n        frequencies=signal_column_frequencies,\n        channel_names=channel_names\n    )\n","metadata":{"execution":{"iopub.status.busy":"2024-02-05T08:33:54.725190Z","iopub.execute_input":"2024-02-05T08:33:54.726675Z","iopub.status.idle":"2024-02-05T08:34:05.250887Z","shell.execute_reply.started":"2024-02-05T08:33:54.726622Z","shell.execute_reply":"2024-02-05T08:34:05.249238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Signal to Spectrogram","metadata":{}},{"cell_type":"code","source":"# To Be Continued","metadata":{},"execution_count":null,"outputs":[]}]}