{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Import necessary Libraries\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport mne\nimport glob\nimport seaborn as sns\nfrom scipy.signal import butter, filtfilt, welch\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, classification_report,roc_curve, auc,log_loss,confusion_matrix,classification_report\nfrom scipy.fft import fft\nfrom sklearn.ensemble import RandomForestClassifier\nimport pywt\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:49:42.384873Z","iopub.execute_input":"2024-04-05T15:49:42.385733Z","iopub.status.idle":"2024-04-05T15:50:03.554770Z","shell.execute_reply.started":"2024-04-05T15:49:42.385684Z","shell.execute_reply":"2024-04-05T15:50:03.552471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This Python script given below is set up to run within a Kaggle environment, which provides a Jupyter notebook interface with access to a range of data science and machine learning tools. It allows to import necessary libraries, such as NumPy and Pandas, which are commonly used for data manipulation and analysis. Then, it lists all files under the \"/kaggle/input/\" directory, which is where input data files are typically stored within a Kaggle project. This helps users to easily access and load datasets for analysis or model training. Additionally, it mentions that the environment allows writing up to 20GB of data to the current directory (\"/kaggle/working/\") for output, and temporary files can be written to \"/kaggle/temp/\" during the session, but they won't persist beyond that session. Overall, this script serves as a starting point for data analysis or machine learning tasks within the Kaggle platform.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-05T15:50:03.556971Z","iopub.execute_input":"2024-04-05T15:50:03.557796Z","iopub.status.idle":"2024-04-05T15:50:39.198170Z","shell.execute_reply.started":"2024-04-05T15:50:03.557750Z","shell.execute_reply":"2024-04-05T15:50:39.197062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"eeg_data.head()","metadata":{}},{"cell_type":"code","source":"# Load the traing data saved in .csv file format\ndf=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:50:39.199498Z","iopub.execute_input":"2024-04-05T15:50:39.199921Z","iopub.status.idle":"2024-04-05T15:50:39.403560Z","shell.execute_reply.started":"2024-04-05T15:50:39.199883Z","shell.execute_reply":"2024-04-05T15:50:39.402110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create dataFrame containing EEG data and groups it by the column 'expert_consensus'. For each unique value in 'expert_consensus', it creates a subset DataFrame containing only the 'eeg_id' column. Each subset DataFrame is then saved as a CSV file with the name of the unique value from 'expert_consensus' appended to the output folder path. This allows the EEG data to be organized into separate CSV files based on different levels of expert consensus, facilitating further analysis or storage.","metadata":{}},{"cell_type":"code","source":"\n\n# # Output folder for CSV files\ncsv_output_folder = '/kaggle/working/grouped_by_expert_consensus'\n\n# # Create the output folder if it doesn't exist\nos.makedirs(csv_output_folder, exist_ok=True)\n\n# # Group the DataFrame by 'expert_consensus' and iterate through groups\nfor group_name, group_df in df.groupby('expert_consensus'):\n     # Create a new DataFrame with 'eeg_id' and 'eeg' columns\n    group_subset_df = group_df[['eeg_id']]\n    \n     # Save the DataFrame as a CSV file with the name of the unique value appended to the output folder\n    csv_output_path = os.path.join(csv_output_folder, f'{group_name.lower()}_id.csv')\n    group_subset_df.to_csv(csv_output_path, index=False)#import glob\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:50:39.405755Z","iopub.execute_input":"2024-04-05T15:50:39.406500Z","iopub.status.idle":"2024-04-05T15:50:39.567879Z","shell.execute_reply.started":"2024-04-05T15:50:39.406459Z","shell.execute_reply":"2024-04-05T15:50:39.566273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### glob module is used to find all the files with the extension \".parquet\" in the directory \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/\", and stores the list of file paths in the variable parquet_files","metadata":{}},{"cell_type":"code","source":"\nparquet_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:50:39.569904Z","iopub.execute_input":"2024-04-05T15:50:39.570377Z","iopub.status.idle":"2024-04-05T15:50:39.619288Z","shell.execute_reply.started":"2024-04-05T15:50:39.570333Z","shell.execute_reply":"2024-04-05T15:50:39.617322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(parquet_files))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T23:13:06.748028Z","iopub.execute_input":"2024-03-11T23:13:06.748343Z","iopub.status.idle":"2024-03-11T23:13:06.752401Z","shell.execute_reply.started":"2024-03-11T23:13:06.748318Z","shell.execute_reply":"2024-03-11T23:13:06.751853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The csv file created above for different group ID is read in the form of data frame.","metadata":{}},{"cell_type":"code","source":"\ndf1=pd.read_csv('/kaggle/working/grouped_by_expert_consensus/seizure_id.csv')\ndf2=pd.read_csv('/kaggle/working/grouped_by_expert_consensus/lpd_id.csv')\ndf3=pd.read_csv('/kaggle/working/grouped_by_expert_consensus/gpd_id.csv')\ndf4=pd.read_csv('/kaggle/working/grouped_by_expert_consensus/lrda_id.csv')\ndf5=pd.read_csv('/kaggle/working//grouped_by_expert_consensus/grda_id.csv')\ndf6=pd.read_csv('/kaggle/working//grouped_by_expert_consensus/other_id.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:50:39.621786Z","iopub.execute_input":"2024-04-05T15:50:39.622353Z","iopub.status.idle":"2024-04-05T15:50:39.652998Z","shell.execute_reply.started":"2024-04-05T15:50:39.622301Z","shell.execute_reply":"2024-04-05T15:50:39.651319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### These code extract EEG ID numbers from parquet file names, then create lists of file paths for seizure files based on matching EEG ID numbers in different DataFrames (df1, df2, df3, df4, df5, df6)","metadata":{}},{"cell_type":"code","source":"#Extracting EEG ID numbers from the parquet file names\neeg_id_numbers = [int(file.split('/')[-1].split('.')[0]) for file in parquet_files]\n\n#Creating seizure file paths based on matching EEG ID numbers\nseizure_file_path = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df1['eeg_id'].tolist()]\n\n#Print the resulting seizure file paths\n#print(seizure_file_path)\n# Creating lpd file paths based on matching EEG ID numbers for df2\nlpd_file_path_df2 = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df2['eeg_id'].tolist()]\n# Creating gpd file paths based on matching EEG ID numbers for df3\ngpd_file_path_df3 = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df3['eeg_id'].tolist()]\n# Creating lrda file paths based on matching EEG ID numbers for df4\nlrda_file_path_df4 = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df4['eeg_id'].tolist()]\n# Creating grda file paths based on matching EEG ID numbers for df5\ngrda_file_path_df5 = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df5['eeg_id'].tolist()]\n# Creating other file paths based on matching EEG ID numbers for df6\nother_file_path_df6 = [file_path for file_path, eeg_id in zip(parquet_files, eeg_id_numbers) if eeg_id in df6['eeg_id'].tolist()]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:50:39.654869Z","iopub.execute_input":"2024-04-05T15:50:39.655326Z","iopub.status.idle":"2024-04-05T15:51:37.193828Z","shell.execute_reply.started":"2024-04-05T15:50:39.655281Z","shell.execute_reply":"2024-04-05T15:51:37.191269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Relabing the files based on the group made using expert consensus.","metadata":{}},{"cell_type":"code","source":"#EEG data are labelled based on the group consensus\nseizure_data = [pd.read_parquet(file_path) for file_path in seizure_file_path]\n\n#Now, seizure_data is a list of DataFrames, each containing the data from a seizure file\n#You can access individual DataFrames using indices, for example:\n#first_seizure_df = seizure_data[1000]\nfor df in seizure_data:\n    df['label'] = 1\n\n#Now, each DataFrame in seizure_data has a new column 'label' with the value 1\n#You can access individual labeled DataFrames using indices, for example:\n#first_seizure_df = seizure_data[3000]\n# Read parquet files corresponding to lpd_file_path\nlpd_data = [pd.read_parquet(file_path) for file_path in lpd_file_path_df2]\n\n# Now, lpd_data is a list of DataFrames, each containing the data from a lpd file\n# You can access individual DataFrames using indices, for example:\n#first_lpd_df = lpd_data[1000]\nfor dfl in lpd_data:\n    dfl['label'] = 2\n\n\n# You can access individual labeled DataFrames using indices, for example:\n#first_lpd_df = lpd_data[300]\n# Read parquet files corresponding to gpd_file_path\ngpd_data = [pd.read_parquet(file_path) for file_path in gpd_file_path_df3]\n\n# Now, gpd_data is a list of DataFrames, each containing the data from a gpd file\n# You can access individual DataFrames using indices, for example:\n#first_gpd_df = gpd_data[1000]\nfor dfg in gpd_data:\n    dfg['label'] = 3\n\n\n# You can access individual labeled DataFrames using indices, for example:\nfirst_gpd_df = gpd_data[300]\n# Read parquet files corresponding to lrda_file_path\nlrda_data = [pd.read_parquet(file_path) for file_path in lrda_file_path_df4]\n\n\n# You can access individual DataFrames using indices, for example:\n#first_lrda_df = lrda_data[1000]\nfor dflr in lrda_data:\n    dflr['label'] = 4\n#\n\n# You can access individual labeled DataFrames using indices, for example:\n#first_lrda_df = lrda_data[300]\n# Read parquet files corresponding to grda_file_path\ngrda_data = [pd.read_parquet(file_path) for file_path in grda_file_path_df5]\n\nfor dfgr in grda_data:\n    dfgr['label'] = 5\n\n\n# You can access individual labeled DataFrames using indices, for example:\n#first_gpd_df = grda_data[300]\n# Read parquet files corresponding to other_file_path\nother_data = [pd.read_parquet(file_path) for file_path in other_file_path_df6]\nfor dfother in other_data:\n    dfother['label'] = 6\n\n\n# You can access individual DataFrames using indices, for example:\nother_grda_df = other_data[1000]","metadata":{"execution":{"iopub.status.busy":"2024-04-05T15:51:37.196259Z","iopub.execute_input":"2024-04-05T15:51:37.196768Z","iopub.status.idle":"2024-04-05T16:04:44.854600Z","shell.execute_reply.started":"2024-04-05T15:51:37.196721Z","shell.execute_reply":"2024-04-05T16:04:44.851438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Due to memory issu I only analize few sample, but one with efficent memory can use all the parquet files for analysizing just removing the indexing term. So 15 parquet files from each category were chosen and store in the corresponding csv file.","metadata":{}},{"cell_type":"code","source":"concatenated_df = pd.concat(seizure_data[:15], ignore_index=True)\nconcatenated_df.to_csv('Seizure_data.csv', index=False)\nconcatenated_lpddf = pd.concat(lpd_data[:15], ignore_index=True)\nconcatenated_lpddf.to_csv('lpd_data.csv', index=False)\nconcatenated_lrdadf = pd.concat(lrda_data[:15], ignore_index=True)\nconcatenated_lpddf = pd.concat(gpd_data[:15], ignore_index=True)\nconcatenated_lpddf.to_csv('gpd_data.csv', index=False)\nconcatenated_lrdadf = pd.concat(lrda_data[:15], ignore_index=True)\nconcatenated_lrdadf.to_csv('lrda_data.csv', index=False)\nconcatenated_grdadf = pd.concat(grda_data[:15], ignore_index=True)\nconcatenated_grdadf.to_csv('grda_data.csv', index=False)\nconcatenated_other = pd.concat(other_data[:15], ignore_index=True)\nconcatenated_other.to_csv('other_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:04:44.859911Z","iopub.execute_input":"2024-04-05T16:04:44.860474Z","iopub.status.idle":"2024-04-05T16:05:10.876118Z","shell.execute_reply.started":"2024-04-05T16:04:44.860413Z","shell.execute_reply":"2024-04-05T16:05:10.873476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CSV file of catual EEG data for different groups are read as data frame.","metadata":{}},{"cell_type":"code","source":"\n\nseizure_data=pd.read_csv('Seizure_data.csv')\nlpd_data=pd.read_csv('lpd_data.csv')\ngpd_data =pd.read_csv('gpd_data.csv')\nlrda_data =pd.read_csv('lrda_data.csv')\ngrda_data =pd.read_csv('grda_data.csv')\nother_data=pd.read_csv('other_data.csv')\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:05:10.881384Z","iopub.execute_input":"2024-04-05T16:05:10.882732Z","iopub.status.idle":"2024-04-05T16:05:14.957663Z","shell.execute_reply.started":"2024-04-05T16:05:10.882681Z","shell.execute_reply":"2024-04-05T16:05:14.956395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n# import matplotlib.pyplot as plt\n# import seaborn as sns\n# import mne\n\n# # Load EEG data (replace 'your_eeg_data.csv' with your actual data file)\n# data = pd.read_csv('Seizure_data.csv')\n\n# # Display basic information about the data\n# print(data.info())\n\n# # Summary statistics\n# print(data.describe())\n\n# # Visualize the first few rows of the data\n# print(data.head())\n\n# # Check for missing values\n# missing_values = data.isnull().sum()\n# print(\"Missing Values:\\n\", missing_values)\n\n# # Visualize the distribution of EEG signal amplitudes for a sample channel\n# channel_name = 'Fz'  # Replace with the desired channel name\n# plt.figure(figsize=(12, 6))\n# sns.histplot(data[channel_name], bins=50, kde=True, color='skyblue')\n# plt.title(f'Distribution of EEG Signal Amplitudes for Channel {channel_name}')\n# plt.xlabel('Amplitude')\n# plt.ylabel('Frequency')\n# plt.show()\n\n# # Plot EEG signals for multiple channels\n# selected_channels = ['Fz', 'Cz', 'Pz']  # Replace with your desired channel names\n# plt.figure(figsize=(12, 6))\n# data[selected_channels].plot(subplots=True)\n# plt.suptitle('EEG Signals for Selected Channels')\n# plt.show()\n\n# # Boxplot of EEG signal amplitudes across channels\n# plt.figure(figsize=(12, 6))\n# sns.boxplot(data=data[selected_channels])\n# plt.title('Boxplot of EEG Signal Amplitudes Across Channels')\n# plt.ylabel('Amplitude')\n# plt.show()\n\n# # Explore temporal patterns using line plots\n# plt.figure(figsize=(12, 6))\n# data[selected_channels].plot()\n# plt.title('Temporal Patterns of EEG Signals')\n# plt.xlabel('Sample')\n# plt.ylabel('Amplitude')\n# plt.show()\n\n# # Plot a heatmap to visualize correlations between channels\n# correlation_matrix = data[selected_channels].corr()\n# plt.figure(figsize=(10, 8))\n# sns.heatmap(correlation_matrix, cmap='coolwarm', annot=True, fmt=\".2f\", linewidths=.5)\n# plt.title('Correlation Matrix of Selected EEG Channels')\n# plt.show()\n\n# # Plot power spectral density (PSD) for selected channels\n# sfreq = 256  # Replace with the actual sampling frequency of your data\n# plt.figure(figsize=(12, 8))\n# for channel in selected_channels:\n#     plt.psd(data[channel], Fs=sfreq, NFFT=1024, label=channel)\n\n# plt.title('Power Spectral Density (PSD) of Selected EEG Channels')\n# plt.xlabel('Frequency (Hz)')\n# plt.ylabel('Power/Frequency (dB/Hz)')\n# plt.legend()\n# plt.show()\n\n# # Create an MNE Raw object for further analysis and visualization\n# info = mne.create_info(ch_names=data.columns, sfreq=sfreq, ch_types='eeg')\n# raw = mne.io.RawArray(data.values.T, info)\n\n# # Plot topomap for a specific time point (adjust time_point variable)\n# time_point = 5  # Replace with the desired time point in seconds\n# raw.plot_topomap(times=[time_point], ch_type='eeg', size=3, title=f'Topomap at {time_point} seconds')\n# plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We defines functions to preprocess EEG data stored in a pandas DataFrame. The butter_bandpass and butter_bandpass_filter functions are used to design and apply a bandpass filter to the EEG data. The process_eeg_data function then applies various preprocessing steps, including filtering, notch filtering to remove power line interference, removing unwanted channels, applying Independent Component Analysis (ICA) for artifact removal, dividing the data into epochs, rejecting epochs with artifacts, computing Power Spectral Density (PSD) estimates, interpolating bad channels, and visualizing EEG data from the first epoch. The function returns a MNE-Python Epochs object containing the preprocessed EEG data, EEG data from the first epoch, PSD estimates, and corresponding frequencies.","metadata":{}},{"cell_type":"code","source":"\n\ndef butter_bandpass(lowcut, highcut, fs, order=4):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    return b, a\n\ndef butter_bandpass_filter(data, lowcut, highcut, fs, order=4):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = filtfilt(b, a, data)\n    return y\n\ndef process_eeg_data(data_frame):\n    \"\"\"\n    Process EEG data.\n\n    Parameters:\n        data_frame (pd.DataFrame): EEG data in DataFrame format.\n\n    Returns:\n        mne.Epochs: MNE-Python Epochs object.\n        np.ndarray: EEG data from the first epoch.\n        np.ndarray: Power Spectral Density (PSD) estimates.\n        np.ndarray: Corresponding frequencies.\n    \"\"\"\n    # Apply bandpass filter to EEG data\n    fs = 250  # Sampling frequency\n    lowcut = 1.0  # Lower cutoff frequency\n    highcut = 50.0  # Upper cutoff frequency\n    data_frame.dropna(inplace=True)\n    data_frame_filtered = data_frame.apply(lambda x: butter_bandpass_filter(x, lowcut, highcut, fs), axis=0)\n    \n    # Load EEG data into a Raw object\n    data_frame_filtered_transposed = data_frame_filtered.T\n    info = mne.create_info(ch_names=data_frame_filtered.columns.tolist(), sfreq=fs, ch_types='eeg')\n    raw = mne.io.RawArray(data_frame_filtered_transposed.values, info)\n\n    # Notch filter to remove power line interference (e.g., 50 Hz)\n    raw.notch_filter(freqs=[50], picks='eeg')  # Adjust frequency as needed\n\n    raw.drop_channels(['EKG', 'label'])\n\n    montage = mne.channels.make_standard_montage('standard_1005')\n    raw.set_montage(montage, on_missing='ignore')\n\n    # Apply ICA for artifact removal\n    ica = mne.preprocessing.ICA(n_components=len(data_frame.columns)-2, random_state=97, max_iter=800)\n    ica.fit(raw)\n    ica.exclude = [0, 1, 2, 3, 4, 5]  # Example muscle components\n    ica.apply(raw)\n\n    # Create epochs\n    epochs = mne.make_fixed_length_epochs(raw, duration=5, overlap=1)\n\n    # Optional preprocessing steps\n    # 1. Artifact rejection\n    epochs.drop_bad()\n    \n    # Compute PSD estimates for each epoch using Welch's method\n    psds = []\n    freqs = None\n    for epoch_data in epochs.get_data():\n        f, p = welch(epoch_data.squeeze(), fs=fs, nperseg=256)\n        if freqs is None:\n            freqs = f\n        psds.append(p)\n    psds = np.array(psds)\n\n    # 7. Interpolation of bad channels\n    raw.interpolate_bads()\n\n    # Visualize EEG data from the first epoch\n    if len(epochs) > 0:\n        eeg_data_epoch = epochs.get_data()\n        # Optional: Plot EEG data from the first epoch\n        times = epochs.times\n        plt.figure(figsize=(10, 5))\n        plt.plot(times, eeg_data_epoch[0].T)\n        plt.title('EEG Data - First Epoch')\n        plt.xlabel('Time (s)')\n        plt.ylabel('EEG Amplitude (uV)')\n        plt.show()\n\n        # Optional: Plot drop log\n        epochs.plot_drop_log()\n    else:\n        print(\"No remaining epochs after rejection\")\n\n    return epochs, eeg_data_epoch, psds, freqs","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:05:14.960475Z","iopub.execute_input":"2024-04-05T16:05:14.961171Z","iopub.status.idle":"2024-04-05T16:05:14.983346Z","shell.execute_reply.started":"2024-04-05T16:05:14.961123Z","shell.execute_reply":"2024-04-05T16:05:14.981621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In the following code we call the process_eeg_data function with different sets of EEG data (seizure_data, lpd_data, gpd_data, lrda_data, grda_data, other_data) to preprocess each set separately. For each set of data, the function returns processed epochs, EEG data from the first epoch, PSD estimates, and corresponding frequencies. These results are stored in separate variables prefixed with processed_epochs_, eeg_data_epoch_, psd_, and freqs_ for each dataset, allowing for further analysis and visualization of the preprocessed EEG data.","metadata":{}},{"cell_type":"code","source":"processed_epochs_sizure, eeg_data_epoch_seizure,psd_seizure,freqs_seizure = process_eeg_data(seizure_data)\n\nprocessed_epochs_lpd, eeg_data_epoch_lpd,psd_lpd,freqs_lpd = process_eeg_data(lpd_data)\nprocessed_epochs_gpd, eeg_data_epoch_gpd,psd_gpd,freqs_gpd = process_eeg_data(gpd_data)\nprocessed_epochs_lrda, eeg_data_epoch_lrda,psd_lrda,freqs_lrda = process_eeg_data(lrda_data)\nprocessed_epochs_grda, eeg_data_epoch_grda ,psd_grda,freqs_grda= process_eeg_data(grda_data)\nprocessed_epochs_other, eeg_data_epoch_other ,psd_other,freqs_other= process_eeg_data(other_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:05:14.985240Z","iopub.execute_input":"2024-04-05T16:05:14.985830Z","iopub.status.idle":"2024-04-05T16:06:45.167182Z","shell.execute_reply.started":"2024-04-05T16:05:14.985779Z","shell.execute_reply":"2024-04-05T16:06:45.164027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create labels for each epoch\nseizure_epoch_labels = [0] * len(processed_epochs_sizure)\nlpd_epoch_labels = [1] * len(processed_epochs_lpd)\ngpd_epoch_labels = [2] * len(processed_epochs_gpd)\nlrda_epoch_labels = [3] * len(processed_epochs_lrda)\ngrda_epoch_labels = [4] * len(processed_epochs_grda)\nother_epoch_labels = [5] * len(processed_epochs_other)\n\n# Combine labels for all epochs\nall_epoch_labels =np.hstack (\n    seizure_epoch_labels +\n    lpd_epoch_labels +\n    gpd_epoch_labels +\n    lrda_epoch_labels +\n    grda_epoch_labels +\n    other_epoch_labels\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:06:45.169148Z","iopub.execute_input":"2024-04-05T16:06:45.170702Z","iopub.status.idle":"2024-04-05T16:06:45.183312Z","shell.execute_reply.started":"2024-04-05T16:06:45.170652Z","shell.execute_reply":"2024-04-05T16:06:45.181045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EEG data are combine vertically","metadata":{}},{"cell_type":"code","source":"\nstacked_eeg_data = np.vstack([eeg_data_epoch_seizure,\n                              eeg_data_epoch_lpd,\n                              eeg_data_epoch_gpd,\n                              eeg_data_epoch_lrda,\n                              eeg_data_epoch_grda,\n                              eeg_data_epoch_other])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:06:45.185316Z","iopub.execute_input":"2024-04-05T16:06:45.186676Z","iopub.status.idle":"2024-04-05T16:06:45.249792Z","shell.execute_reply.started":"2024-04-05T16:06:45.186567Z","shell.execute_reply":"2024-04-05T16:06:45.248647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now\n# Extract feature from the raw data\nfrom scipy import stats\ndef mean(x):\n    return np.mean(x,axis=-1)\ndef std(x):\n    return np.std(x, axis=-1)\ndef ptp(x):\n    return np.ptp(x, axis=-1)\ndef var(x):\n    return np.var(x,axis=-1)\ndef minim(x):\n    return np.min(x,axis=-1)\ndef maxim(x):\n    return np.max(x,axis=-1)\ndef argminim(x):\n    return np.argmin(x,axis=-1)\ndef argmaxim(x):\n    return np.argmax(x,axis=-1)\ndef sqrt(x):\n    return np.sqrt(np.mean(x ** 2, axis=-1))\ndef abs_diff_sigma(x):\n    return np.sum(np.abs(np.diff(x,axis=-1)),axis=-1)\ndef skew(x):\n    return stats.skew(x,axis=-1)\ndef kurtosis(x):\n    return stats.kurtosis(x,axis=-1)\ndef concatenate_features(x):\n    return np.concatenate((mean(x),std(x),ptp(x),var(x),minim(x),maxim(x),argminim(x),\n argmaxim(x),sqrt(x),abs_diff_sigma(x),skew(x),\n    kurtosis(x)),axis=-1)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T17:48:38.149695Z","iopub.execute_input":"2024-04-03T17:48:38.150170Z","iopub.status.idle":"2024-04-03T17:48:38.163521Z","shell.execute_reply.started":"2024-04-03T17:48:38.150122Z","shell.execute_reply":"2024-04-03T17:48:38.162510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features=[]\nfor d in stacked_eeg_data:\n    features.append(concatenate_features(d))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T17:48:38.167237Z","iopub.execute_input":"2024-04-03T17:48:38.168469Z","iopub.status.idle":"2024-04-03T17:48:45.556553Z","shell.execute_reply.started":"2024-04-03T17:48:38.168428Z","shell.execute_reply":"2024-04-03T17:48:45.555531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_eeg_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-03T17:52:51.386852Z","iopub.execute_input":"2024-04-03T17:52:51.387432Z","iopub.status.idle":"2024-04-03T17:52:51.396474Z","shell.execute_reply.started":"2024-04-03T17:52:51.387381Z","shell.execute_reply":"2024-04-03T17:52:51.395267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_array=np.array(features)\nfeatures_array.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-03T17:48:45.558108Z","iopub.execute_input":"2024-04-03T17:48:45.559263Z","iopub.status.idle":"2024-04-03T17:48:45.570314Z","shell.execute_reply.started":"2024-04-03T17:48:45.559225Z","shell.execute_reply":"2024-04-03T17:48:45.569216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random forest classifer were used for modeling the extracted features.","metadata":{}},{"cell_type":"code","source":"\n# Flatten the EEG data to 2D\nflattened_eeg_data = features_array.reshape(stacked_eeg_data.shape[0], -1)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(flattened_eeg_data, all_epoch_labels, test_size=0.2, random_state=42)\n\n# Initialize and train a classifier\nclassifier = RandomForestClassifier()\nclassifier.fit(X_train, y_train)\n\n# Make predictions\ny_pred = classifier.predict(X_test)\n\n# Evaluate the model\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy}\")\n\n# Display additional evaluation metrics\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T21:12:58.761343Z","iopub.execute_input":"2024-04-03T21:12:58.761889Z","iopub.status.idle":"2024-04-03T21:13:00.445705Z","shell.execute_reply.started":"2024-04-03T21:12:58.761846Z","shell.execute_reply":"2024-04-03T21:13:00.444510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Evaluate the model\ny_pred_proba = classifier.predict_proba(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nlogloss = log_loss(y_test, y_pred_proba)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(f\"Log Loss: {logloss}\")\n\n# Display additional evaluation metrics\nprint(classification_report(y_test, y_pred))\n\n# Display predicted probabilities\nprint(\"Predicted Probabilities:\")\nprint(y_pred_proba)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T21:13:04.310682Z","iopub.execute_input":"2024-04-03T21:13:04.311234Z","iopub.status.idle":"2024-04-03T21:13:04.352859Z","shell.execute_reply.started":"2024-04-03T21:13:04.311187Z","shell.execute_reply":"2024-04-03T21:13:04.351251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# from sklearn.metrics import roc_curve, auc\n# import numpy as np\n# import tensorflow as tf\n# from sklearn.model_selection import train_test_split\n# from sklearn.preprocessing import StandardScaler\n# from sklearn.metrics import accuracy_score, classification_report\n# from scipy.fft import fft\n# # Flatten the EEG data to 2D\n# #flattened_eeg_data = stacked_eeg_data.reshape(1580, -1)\n# # Apply Fourier transform to each channel\n# reshaped_data = stacked_eeg_data.reshape((stacked_eeg_data.shape[0], -1))\n# fourier_transformed_data = np.abs(fft(reshaped_data, axis=1))\n# fourier_transformed_data = fourier_transformed_data.reshape((stacked_eeg_data.shape[0], -1))\n# # Apply standard scaling\n# scaler = StandardScaler()\n# scaled_data = scaler.fit_transform(fourier_transformed_data)\n\n# # Reshape the data back to the original shape\n# scaled_data = scaled_data.reshape((stacked_eeg_data.shape[0], stacked_eeg_data.shape[1], stacked_eeg_data.shape[2]))\n\n# # # Standardize the data\n# # scaler = StandardScaler()\n# # flattened_eeg_data_scaled = scaler.fit_transform(flattened_eeg_data)\n\n# # Split the data into training and testing sets\n# X_train, X_test, y_train, y_test = train_test_split(scaled_data, all_epoch_labels, test_size=0.2, random_state=42)\n\n# #X_train.shape\n# # # Reshape data for LSTM input (samples, timesteps, features)\n# X_train_reshaped = X_train.reshape((X_train.shape[0], stacked_eeg_data.shape[2], stacked_eeg_data.shape[1]))\n# X_test_reshaped = X_test.reshape((X_test.shape[0],stacked_eeg_data.shape[2], stacked_eeg_data.shape[1]))\n\n# # Build the LSTM model\n# model = tf.keras.Sequential([\n#     tf.keras.layers.LSTM(64, input_shape=(stacked_eeg_data.shape[2], stacked_eeg_data.shape[1]), activation='relu', return_sequences=True),\n#     tf.keras.layers.BatchNormalization(axis=1),\n#     tf.keras.layers.LSTM(64, activation='relu', return_sequences=True),\n#     tf.keras.layers.BatchNormalization(axis=1),\n#     tf.keras.layers.LSTM(64, activation='relu'),\n#     tf.keras.layers.BatchNormalization(axis=1),\n#     tf.keras.layers.Dense(128, activation='relu'),\n#     tf.keras.layers.BatchNormalization(axis=1),\n#     tf.keras.layers.Dense(6, activation='softmax')  # Assuming 6 classes (0 to 5)\n# ])\n\n# # Compile the model\n# model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# # Train the model\n# model.fit(X_train_reshaped, y_train, epochs=10, batch_size=32, validation_data=(X_test_reshaped, y_test))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T17:54:18.490082Z","iopub.execute_input":"2024-04-03T17:54:18.490524Z","iopub.status.idle":"2024-04-03T18:14:52.805028Z","shell.execute_reply.started":"2024-04-03T17:54:18.490489Z","shell.execute_reply":"2024-04-03T18:14:52.803007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Evaluate the model\n# y_pred_probs = model.predict(X_test_reshaped)\n\n# # Convert to numpy array for consistency with scikit-learn metrics\n# y_test_np = np.array(y_test)\n\n# # Calculate ROC curve and AUC for each class\n# fpr = dict()\n# tpr = dict()\n# roc_auc = dict()\n\n# for i in range(6):  # Assuming 6 classes (0 to 5)\n#     fpr[i], tpr[i], _ = roc_curve(y_test_np == i, y_pred_probs[:, i])\n#     roc_auc[i] = auc(fpr[i], tpr[i])\n\n# # Plot ROC curve\n# plt.figure(figsize=(8, 6))\n\n# for i in range(6):  # Assuming 6 classes (0 to 5)\n#     plt.plot(fpr[i], tpr[i], label=f'Class {i} (AUC = {roc_auc[i]:.2f})')\n\n# plt.plot([0, 1], [0, 1], 'k--', label='Random')\n# plt.xlabel('False Positive Rate')\n# plt.ylabel('True Positive Rate')\n# plt.title('ROC Curve')\n# plt.legend(loc='lower right')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T18:16:15.002453Z","iopub.execute_input":"2024-04-03T18:16:15.003005Z","iopub.status.idle":"2024-04-03T18:16:26.370088Z","shell.execute_reply.started":"2024-04-03T18:16:15.002943Z","shell.execute_reply":"2024-04-03T18:16:26.368257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We define a function denoise_signal that applies wavelet denoising to a one-dimensional signal. The signal is decomposed using the discrete wavelet transform (DWT) into approximation and detail coefficients at the specified level. Then, a threshold is estimated based on the mean absolute deviation (MAD) of the detail coefficients, and soft thresholding is applied to shrink the coefficients below the threshold. Finally, the denoised signal is reconstructed using the inverse wavelet transform.\n\n### The denoising function is applied to a stacked EEG dataset, where each epoch contains EEG data from multiple channels. The denoised signal is stored in a new array denoised_stacked_eeg_data. The code iterates over each epoch and each channel within the epoch, applying denoising to each channel's EEG data separately. The original and denoised EEG signals are then plotted for visualization, demonstrating the effect of wavelet denoising on the EEG data.","metadata":{"execution":{"iopub.status.busy":"2024-04-03T19:44:13.426809Z","iopub.execute_input":"2024-04-03T19:44:13.427472Z","iopub.status.idle":"2024-04-03T19:44:13.438152Z","shell.execute_reply.started":"2024-04-03T19:44:13.427421Z","shell.execute_reply":"2024-04-03T19:44:13.436651Z"}}},{"cell_type":"code","source":"def denoise_signal(signal, wavelet='db4', level=5):\n    # Decompose the signal using DWT\n    coeffs = pywt.wavedec(signal, wavelet, level=level, axis=0)  # Decompose along the first axis\n    \n    # Estimate threshold based on the mean absolute deviation (MAD)\n    sigma = np.median(np.abs(coeffs[-1])) / 0.6745\n    threshold = sigma * np.sqrt(2 * np.log(len(signal)))\n    \n    # Thresholding the detail coefficients\n    coeffs[1:] = tuple(pywt.threshold(c, value=threshold, mode='soft') for c in coeffs[1:])\n    \n    # Reconstruct the denoised signal\n    denoised_signal = pywt.waverec(coeffs, wavelet, axis=0)\n    return denoised_signal[:len(signal)]  # Trim the denoised signal to match the original length\n\n# Apply wavelet denoising to the stacked EEG data\ndenoised_stacked_eeg_data = np.zeros_like(stacked_eeg_data)\nfor i in range(stacked_eeg_data.shape[0]):  # Iterate over epoch\n    for j in range(stacked_eeg_data.shape[1]):  # Iterate over channel\n        denoised_stacked_eeg_data[i, j,:] = denoise_signal(stacked_eeg_data[i, j,:])\n\n# Plotting original and denoised EEG signals for visualization\nplt.figure(figsize=(10, 6))\nplt.plot(stacked_eeg_data[:, 0, 0], label='Original EEG Signal')\nplt.plot(denoised_stacked_eeg_data[:, 0, 0], label='Denoised EEG Signal')\nplt.title('EEG Signal Denoising with Wavelet Transform')\nplt.xlabel('Time (samples)')\nplt.ylabel('Amplitude')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T19:53:50.508934Z","iopub.execute_input":"2024-04-03T19:53:50.509599Z","iopub.status.idle":"2024-04-03T19:54:17.829742Z","shell.execute_reply.started":"2024-04-03T19:53:50.509547Z","shell.execute_reply":"2024-04-03T19:54:17.828274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training a Long Short-Term Memory (LSTM) neural network model for EEG signal classification. It begins by splitting the denoised EEG signals and their corresponding labels into training, validation, and test sets. The data is reshaped to fit the input requirements of the LSTM model. The LSTM architecture is defined with 64 units and an input shape corresponding to the dimensions of the training data. The model is compiled with the Adam optimizer and sparse categorical cross-entropy loss function. Then, it's trained on the training data for 10 epochs with a batch size of 32, using the validation set for validation during training. Finally, the model's performance is evaluated on the test set, and the test accuracy is printed.","metadata":{}},{"cell_type":"code","source":"\n\n# Assuming you have denoised EEG signals (denoised_stacked_eeg_data) and corresponding labels (labels)\n# You may need to reshape denoised EEG signals if necessary\n\n# Split data into training, validation, and test sets\nX_train, X_temp, y_train, y_temp = train_test_split(denoised_stacked_eeg_data, all_epoch_labels, test_size=0.3, random_state=42)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, random_state=42)\n\nX_train_reshaped = np.transpose(X_train, (0, 2, 1))\n\n# Similarly reshape X_val and X_test\nX_val_reshaped = np.transpose(X_val, (0, 2, 1))\nX_test_reshaped = np.transpose(X_test, (0, 2, 1))\n\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, Dense\n\n# Define LSTM and GRU model architecture\nmodel = Sequential()\nmodel.add(LSTM(units=64, return_sequences=True, input_shape=(X_train.shape[2], X_train.shape[1])))  # LSTM layer\nmodel.add(LSTM(units=64, return_sequences=True))  # Second LSTM layer\nmodel.add(GRU(units=64, return_sequences=True))  # GRU layer\nmodel.add(LSTM(units=64))  # Third LSTM layer\nmodel.add(Dense(6, activation='softmax'))\n\n# Compile model\nmodel.compile(loss='sparse_categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# Train model\nhistory = model.fit(X_train_reshaped, y_train, epochs=10, batch_size=32, validation_data=(X_val_reshaped, y_val))\n\n# Evaluate model\ntest_loss, test_acc = model.evaluate(X_test_reshaped, y_test)\nprint(\"Test accuracy:\", test_acc)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T19:54:32.887616Z","iopub.execute_input":"2024-04-03T19:54:32.888176Z","iopub.status.idle":"2024-04-03T19:59:33.590218Z","shell.execute_reply.started":"2024-04-03T19:54:32.888133Z","shell.execute_reply":"2024-04-03T19:59:33.588819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### It performs the evaluation of a classification model using several metrics. First, it calculates the confusion matrix to visualize the model's performance in terms of the true positives, true negatives, false positives, and false negatives for each class. Then, it calculates and plots the Receiver Operating Characteristic (ROC) curve and Area Under the Curve (AUC) for each class, along with the overall ROC curve. Finally, it generates a classification report, which includes metrics such as precision, recall, F1-score, and support for each class, providing a comprehensive summary of the model's performance across different classes","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Function to calculate sensitivity (true positive rate) for each class\ndef sensitivity(conf_matrix):\n    return np.diag(conf_matrix) / np.sum(conf_matrix, axis=1)\n\n# Evaluate the model\ny_pred_probs = model.predict(X_test_reshaped)\ny_pred = np.argmax(y_pred_probs, axis=1)\ny_test_np = np.array(y_test)\n\n# Calculate confusion matrix\nconf_matrix = confusion_matrix(y_test_np, y_pred)\n\n# Calculate sensitivity for each class\nsen = sensitivity(conf_matrix)\n\n# Print sensitivity for each class\nfor i, s in enumerate(sen):\n    print(f'Sensitivity for Class {i}: {s:.2f}')\n\n# Plot confusion matrix\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\n\n# Compute ROC curve and ROC area for each class\nn_classes = len(np.unique(y_test_np))\n\nplt.subplot(1, 2, 2)\nfor i in range(n_classes):\n    fpr, tpr, _ = roc_curve((y_test_np == i).astype(int), y_pred_probs[:, i])\n    roc_auc = auc(fpr, tpr)\n    plt.plot(fpr, tpr, label=f'Class {i} (AUC = {roc_auc:.2f})')\n\nplt.plot([0, 1], [0, 1], 'k--', label='Random')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc='lower right')\n\nplt.tight_layout()\nplt.show()\n\n# Generate classification report\nclass_report = classification_report(y_test_np, y_pred, target_names=[f'Class {i}' for i in range(n_classes)])\n\n# Print classification report\nprint(class_report)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T21:33:36.894753Z","iopub.execute_input":"2024-04-03T21:33:36.895581Z","iopub.status.idle":"2024-04-03T21:33:37.738157Z","shell.execute_reply.started":"2024-04-03T21:33:36.895514Z","shell.execute_reply":"2024-04-03T21:33:37.736733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/3911565283.parquet')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocssing for test data","metadata":{}},{"cell_type":"code","source":"def process_eeg_datatest(data_frame):\n    \"\"\"\n    Process EEG data.\n\n    Parameters:\n        data_frame (pd.DataFrame): EEG data in DataFrame format.\n\n    Returns:\n        mne.Epochs: MNE-Python Epochs object.\n        np.ndarray: EEG data from the first epoch.\n        np.ndarray: Power Spectral Density (PSD) estimates.\n        np.ndarray: Corresponding frequencies.\n    \"\"\"\n    # Apply bandpass filter to EEG data\n    fs = 250  # Sampling frequency\n    lowcut = 1.0  # Lower cutoff frequency\n    highcut = 50.0  # Upper cutoff frequency\n    data_frame.dropna(inplace=True)\n    data_frame_filtered = data_frame.apply(lambda x: butter_bandpass_filter(x, lowcut, highcut, fs), axis=0)\n    \n    # Load EEG data into a Raw object\n    data_frame_filtered_transposed = data_frame_filtered.T\n    info = mne.create_info(ch_names=data_frame_filtered.columns.tolist(), sfreq=fs, ch_types='eeg')\n    raw = mne.io.RawArray(data_frame_filtered_transposed.values, info)\n\n    # Notch filter to remove power line interference (e.g., 50 Hz)\n    raw.notch_filter(freqs=[50], picks='eeg')  # Adjust frequency as needed\n\n    raw.drop_channels(['EKG'])\n\n    montage = mne.channels.make_standard_montage('standard_1005')\n    raw.set_montage(montage, on_missing='ignore')\n\n    # Apply ICA for artifact removal\n    ica = mne.preprocessing.ICA(n_components=len(data_frame.columns)-2, random_state=97, max_iter=800)\n    ica.fit(raw)\n    ica.exclude = [0, 1, 2, 3, 4, 5]  # Example muscle components\n    ica.apply(raw)\n\n    # Create epochs\n    epochs = mne.make_fixed_length_epochs(raw, duration=5, overlap=1)\n\n    # Optional preprocessing steps\n    # 1. Artifact rejection\n    epochs.drop_bad()\n    \n    # Compute PSD estimates for each epoch using Welch's method\n    psds = []\n    freqs = None\n    for epoch_data in epochs.get_data():\n        f, p = welch(epoch_data.squeeze(), fs=fs, nperseg=256)\n        if freqs is None:\n            freqs = f\n        psds.append(p)\n    psds = np.array(psds)\n\n    # 7. Interpolation of bad channels\n    raw.interpolate_bads()\n\n    # Visualize EEG data from the first epoch\n    if len(epochs) > 0:\n        eeg_data_epoch = epochs.get_data()\n        # Optional: Plot EEG data from the first epoch\n        times = epochs.times\n        plt.figure(figsize=(10, 5))\n        plt.plot(times, eeg_data_epoch[0].T)\n        plt.title('EEG Data - First Epoch')\n        plt.xlabel('Time (s)')\n        plt.ylabel('EEG Amplitude (uV)')\n        plt.show()\n\n        # Optional: Plot drop log\n        epochs.plot_drop_log()\n    else:\n        print(\"No remaining epochs after rejection\")\n\n    return epochs, eeg_data_epoch, psds, freqs","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:00:46.490537Z","iopub.execute_input":"2024-04-03T20:00:46.491098Z","iopub.status.idle":"2024-04-03T20:00:46.509150Z","shell.execute_reply.started":"2024-04-03T20:00:46.491054Z","shell.execute_reply":"2024-04-03T20:00:46.507308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# assiginging preprocess data \nepochstest, eeg_data_epochtest, psdstest, freqstest = process_eeg_datatest(test)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:18:15.915837Z","iopub.execute_input":"2024-04-03T20:18:15.917156Z","iopub.status.idle":"2024-04-03T20:18:46.201818Z","shell.execute_reply.started":"2024-04-03T20:18:15.917103Z","shell.execute_reply":"2024-04-03T20:18:46.200065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_data_epochtest.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:18:46.205342Z","iopub.execute_input":"2024-04-03T20:18:46.205921Z","iopub.status.idle":"2024-04-03T20:18:46.214890Z","shell.execute_reply.started":"2024-04-03T20:18:46.205869Z","shell.execute_reply":"2024-04-03T20:18:46.213549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- ### Apply wavelet denoising inthe preprocess data and predict the test data using the LSTM model made above. Finally, we save the reuslt in submission.csv -->","metadata":{}},{"cell_type":"code","source":"# Apply wavelet denoising to the stacked EEG data\ndenoised_stacked_eeg_data_test = np.zeros_like(eeg_data_epochtest)\n# Apply wavelet denoising to the stacked EEG data\n#denoised_stacked_eeg_data_test = np.zeros_like(eeg_data_epoch_test)\nfor i in range(eeg_data_epochtest.shape[0]):  # Iterate over epoch\n    for j in range(eeg_data_epochtest.shape[1]):  # Iterate over channel\n        denoised_stacked_eeg_data_test[i, j,:] = denoise_signal(eeg_data_epochtest[i, j,:])\n\n# Predict probabilities using the model\ndenoised_stacked_eeg_data_test = np.transpose(denoised_stacked_eeg_data_test, (0, 2, 1))\npredicted_probabilities = model.predict(denoised_stacked_eeg_data_test)\n#processed_epochs_test, eeg_data_epoch_test = process_eeg_datatest(test)\n#featurestest=[]\n#for d in eeg_data_epoch_test:\n #   featurestest.append(concatenate_features(d))\n#features_arraytest=np.array(featurestest)\n#y_predtest = classifier.predict(features_arraytest[:, :features_array.shape[1]])\n#y_pred_probatest = classifier.predict_proba(features_arraytest[:, :features_array.shape[1]])\n\n#y_predtest = model.predict(eeg_data_epoch_test)\n#row_averages = np.mean(y_pred_probatest, axis=0)\naverage_probabilities = np.mean(predicted_probabilities, axis=0)\ndf=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\n# Update the DataFrame with the predicted probabilities\ndf[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']] = average_probabilities \n# Save the updated DataFrame to a new CSV file\ndf.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:25:00.253545Z","iopub.execute_input":"2024-04-03T20:25:00.254104Z","iopub.status.idle":"2024-04-03T20:25:00.648552Z","shell.execute_reply.started":"2024-04-03T20:25:00.254058Z","shell.execute_reply":"2024-04-03T20:25:00.647371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#denoised_stacked_eeg_data_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:22:10.423443Z","iopub.execute_input":"2024-04-03T20:22:10.424280Z","iopub.status.idle":"2024-04-03T20:22:10.434885Z","shell.execute_reply.started":"2024-04-03T20:22:10.424219Z","shell.execute_reply":"2024-04-03T20:22:10.433101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import pandas as pd\n#df2=pd.read_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:25:19.492680Z","iopub.execute_input":"2024-04-03T20:25:19.493249Z","iopub.status.idle":"2024-04-03T20:25:19.503432Z","shell.execute_reply.started":"2024-04-03T20:25:19.493205Z","shell.execute_reply":"2024-04-03T20:25:19.502073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T20:26:17.560437Z","iopub.execute_input":"2024-04-03T20:26:17.560949Z","iopub.status.idle":"2024-04-03T20:26:17.586423Z","shell.execute_reply.started":"2024-04-03T20:26:17.560906Z","shell.execute_reply":"2024-04-03T20:26:17.585083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install PyEMD","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:12:16.420364Z","iopub.execute_input":"2024-04-05T16:12:16.420988Z","iopub.status.idle":"2024-04-05T16:12:47.293890Z","shell.execute_reply.started":"2024-04-05T16:12:16.420942Z","shell.execute_reply":"2024-04-05T16:12:47.292866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- ### In this code, we are using Empirical Mode Decomposition (EMD) to analyze electroencephalogram (EEG) signals. First, we define functions for EMD, including finding extrema in the signal, linear interpolation, and extracting the mean envelope. Then, we decompose the EEG signal into Intrinsic Mode Functions (IMFs) using the EMD algorithm. Each IMF represents a component of the original signal with a characteristic oscillatory behavior. Next, we apply Hilbert transform to each IMF to obtain its analytic signal, which provides information about the instantaneous amplitude and phase. After that, we design a finite impulse response (FIR) filter to filter each IMF, removing unwanted frequency components. The filtered IMFs are stored in the filtered_imfs list.The overall process involves decomposing the EEG signal into IMFs using EMD, analyzing each IMF using Hilbert transform, and then filtering each IMF to extract specific frequency components. This process allows us to gain insights into the underlying oscillatory patterns present in the EEG signal. -->","metadata":{}},{"cell_type":"code","source":"# import numpy as np\n# import matplotlib.pyplot as plt\n# from scipy.signal import hilbert, firwin, filtfilt\n# #from PyEMD import EMD\n\n# # Assuming stacked_eeg_data is a 3D array where the first dimension represents epochs,\n# # the second dimension represents channels, and the third dimension represents time samples.\n\n# # Define the sampling frequency and time vector\n# sampling_frequency = 200  # Assuming a sampling frequency of 100 Hz\n# time_vector = np.arange(stacked_eeg_data.shape[2]) / sampling_frequency\n\n# # Initialize EMD\n# #emd = EMD()\n\n# # Define filter parameters\n# filter_type = 'bandpass'\n# lowcut = 0.5  # Lower cutoff frequency in Hz\n# highcut = 50  # Upper cutoff frequency in Hz\n# filter_order = 100  # Filter order\n\n# # Design FIR filter\n# nyquist_freq = 0.5 * sampling_frequency\n# if filter_type == 'bandpass':\n#     fir_coeff = firwin(filter_order, [lowcut / nyquist_freq, highcut / nyquist_freq], pass_zero=False)\n# elif filter_type == 'lowpass':\n#     fir_coeff = firwin(filter_order, highcut / nyquist_freq, pass_zero=False)\n# elif filter_type == 'highpass':\n#     fir_coeff = firwin(filter_order, lowcut / nyquist_freq, pass_zero=False)\n\n# import numpy as np\n# import matplotlib.pyplot as plt\n# from scipy.signal import hilbert, firwin, filtfilt\n# from scipy.interpolate import InterpolatedUnivariateSpline\n\n# # Define functions for EMD\n# def find_extrema(signal):\n#     maxima = (signal[:-2] < signal[1:-1]) & (signal[1:-1] > signal[2:])\n#     minima = (signal[:-2] > signal[1:-1]) & (signal[1:-1] < signal[2:])\n#     return maxima, minima\n\n# def linear_interpolate(signal, indices):\n#     x = np.arange(len(signal))\n#     return np.interp(x, indices, signal[indices])\n\n# def extract_mean_envelope(signal):\n#     maxima, minima = find_extrema(signal)\n#     upper_env = linear_interpolate(signal, np.where(maxima)[0])\n#     lower_env = linear_interpolate(signal, np.where(minima)[0])\n#     return (upper_env + lower_env) / 2\n\n\n# def extract_imf(signal, tolerance=0.1, max_iterations=100):\n#     imf = signal.copy()\n#     for _ in range(max_iterations):\n#         mean_env = extract_mean_envelope(imf)\n#         imf_old = imf\n#         imf = imf - mean_env\n#         if np.sum(np.abs(imf_old - imf) / np.abs(imf)) < tolerance:\n#             return imf\n#     return imf\n\n\n# def emd(signal):\n#     imfs = []\n#     residue = signal\n#     while True:\n#         imf = extract_imf(residue)\n#         imfs.append(imf)\n#         residue -= imf\n#         if np.max(np.abs(imf)) < 0.1 * np.max(np.abs(signal)):\n#             imfs.append(residue)\n#             break\n#     return imfs\n\n\n\n# # Decompose the EEG signal for each epoch\n# imfs_all_epochs = []\n# for epoch_data in stacked_eeg_data:\n#     imfs_epoch = []\n#     for channel_data in epoch_data:\n#         imfs = emd(channel_data)\n#          # Apply Hilbert Transform to each IMF\n#         analytic_signals = [hilbert(imf) for imf in imfs]\n        \n#         # Filter each IMF\n#         filtered_imfs = []\n#         for analytic_signal in analytic_signals:\n#             filtered_signal = filtfilt(fir_coeff, 1.0, np.real(analytic_signal))  # Apply filtering\n#             filtered_imfs.append(filtered_signal)\n#         imfs_epoch.append(imfs)\n#     imfs_all_epochs.append(imfs_epoch)\n\n# # # Decompose the EEG signal for each epoch\n# # imfs_all_epochs = []\n# # for epoch_data in stacked_eeg_data[:]:  # Limit to the first two epochs\n# #     imfs_epoch = []\n# #     for channel_data in epoch_data:\n# #         imfs = emd(channel_data)\n        \n# #         # Apply Hilbert Transform to each IMF\n# #         analytic_signals = [hilbert(imf) for imf in imfs]\n        \n# #         # Filter each IMF\n# #         filtered_imfs = []\n# #         for analytic_signal in analytic_signals:\n# #             filtered_signal = filtfilt(fir_coeff, 1.0, np.real(analytic_signal))  # Apply filtering\n# #             filtered_imfs.append(filtered_signal)\n        \n# #         imfs_epoch.append(filtered_imfs)\n# #     imfs_all_epochs.append(imfs_epoch)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T16:44:57.907330Z","iopub.execute_input":"2024-04-05T16:44:57.907920Z","iopub.status.idle":"2024-04-05T17:45:40.664501Z","shell.execute_reply.started":"2024-04-05T16:44:57.907875Z","shell.execute_reply":"2024-04-05T17:45:40.661896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- ### In this section of the code, we're reconstructing the EEG signals after processing them through Empirical Mode Decomposition (EMD) and filtering. We initialize a new variable called reconstructed_signal to store the reconstructed EEG signals. Then, we loop through each epoch of the EEG data. Within each epoch, we iterate through each channel of the EEG data. For each channel, we initialize another variable called reconstructed_channel_signal to store the reconstructed signal for that channel.Next, we iterate through each Intrinsic Mode Function (IMF) obtained from the EMD process. We add up all the IMFs corresponding to that channel to reconstruct the signal for that channel. Finally, we store the reconstructed signal for the current channel and epoch in the reconstructed_signal variable.Overall, this part of the code reconstructs the original EEG signals after decomposing them into simpler components, filtering out noise, and summing up these components to reconstruct the signals for each channel and epoch. -->","metadata":{}},{"cell_type":"code","source":"# reconstructed_signal = np.zeros_like(stacked_eeg_data[:])  # Initialize reconstructed signal\n\n# # Iterate through each epoch\n# for epoch_idx in range(len(imfs_all_epochs)):\n#     # Iterate through each channel\n#     for channel_idx in range(len(imfs_all_epochs[epoch_idx])):\n#         # Initialize reconstructed signal for the current channel\n#         reconstructed_channel_signal = np.zeros_like(stacked_eeg_data[epoch_idx][channel_idx])\n        \n#         # Iterate through each IMF\n#         for imf_idx in range(len(imfs_all_epochs[epoch_idx][channel_idx])):\n#             # Add current IMF to the reconstructed channel signal\n#             reconstructed_channel_signal += imfs_all_epochs[epoch_idx][channel_idx][imf_idx]\n        \n#         # Store the reconstructed signal for the current channel and epoch\n#         reconstructed_signal[epoch_idx][channel_idx] = reconstructed_channel_signal","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:45:40.668782Z","iopub.execute_input":"2024-04-05T17:45:40.669582Z","iopub.status.idle":"2024-04-05T17:45:42.355800Z","shell.execute_reply.started":"2024-04-05T17:45:40.669516Z","shell.execute_reply":"2024-04-05T17:45:42.353780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Assuming you have denoised EEG signals (denoised_stacked_eeg_data) and corresponding labels (labels)\n# # You may need to reshape denoised EEG signals if necessary\n\n# # Split data into training, validation, and test sets\n# X_train, X_temp, y_train, y_temp = train_test_split(reconstructed_signal, all_epoch_labels, test_size=0.3, random_state=42)\n# X_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, random_state=42)\n\n# X_train_reshaped = np.transpose(X_train, (0, 2, 1))\n\n# # Similarly reshape X_val and X_test\n# X_val_reshaped = np.transpose(X_val, (0, 2, 1))\n# X_test_reshaped = np.transpose(X_test, (0, 2, 1))\n\n# # Define LSTM model architecture\n# model = Sequential()\n# model.add(LSTM(units=64, input_shape=(X_train.shape[2], X_train.shape[1])))\n# model.add(Dense(6, activation='softmax'))\n\n# # Compile model\n# model.compile(loss='sparse_categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# # Train model\n# history = model.fit(X_train_reshaped, y_train, epochs=10, batch_size=32, validation_data=(X_val_reshaped, y_val))\n\n# # Evaluate model\n# test_loss, test_acc = model.evaluate(X_test_reshaped, y_test)\n# print(\"Test accuracy:\", test_acc)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:45:42.357692Z","iopub.execute_input":"2024-04-05T17:45:42.358244Z","iopub.status.idle":"2024-04-05T17:49:52.296970Z","shell.execute_reply.started":"2024-04-05T17:45:42.358197Z","shell.execute_reply":"2024-04-05T17:49:52.295283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc\n# import seaborn as sns\n# import matplotlib.pyplot as plt\n# import numpy as np\n\n# # Function to calculate sensitivity (true positive rate) for each class\n# def sensitivity(conf_matrix):\n#     return np.diag(conf_matrix) / np.sum(conf_matrix, axis=1)\n\n# # Evaluate the model\n# y_pred_probs = model.predict(X_test_reshaped)\n# y_pred = np.argmax(y_pred_probs, axis=1)\n# y_test_np = np.array(y_test)\n\n# # Calculate confusion matrix\n# conf_matrix = confusion_matrix(y_test_np, y_pred)\n\n# # Calculate sensitivity for each class\n# sen = sensitivity(conf_matrix)\n\n# # Print sensitivity for each class\n# for i, s in enumerate(sen):\n#     print(f'Sensitivity for Class {i}: {s:.2f}')\n\n# # Plot confusion matrix\n# plt.figure(figsize=(12, 5))\n# plt.subplot(1, 2, 1)\n# sns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\n# plt.xlabel('Predicted Label')\n# plt.ylabel('True Label')\n# plt.title('Confusion Matrix')\n\n# # Compute ROC curve and ROC area for each class\n# n_classes = len(np.unique(y_test_np))\n\n# plt.subplot(1, 2, 2)\n# for i in range(n_classes):\n#     fpr, tpr, _ = roc_curve((y_test_np == i).astype(int), y_pred_probs[:, i])\n#     roc_auc = auc(fpr, tpr)\n#     plt.plot(fpr, tpr, label=f'Class {i} (AUC = {roc_auc:.2f})')\n\n# plt.plot([0, 1], [0, 1], 'k--', label='Random')\n# plt.xlabel('False Positive Rate')\n# plt.ylabel('True Positive Rate')\n# plt.title('Receiver Operating Characteristic (ROC) Curve')\n# plt.legend(loc='lower right')\n\n# plt.tight_layout()\n# plt.show()\n\n# # Generate classification report\n# class_report = classification_report(y_test_np, y_pred, target_names=[f'Class {i}' for i in range(n_classes)])\n\n# # Print classification report\n# print(class_report)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:49:52.299130Z","iopub.execute_input":"2024-04-05T17:49:52.299917Z","iopub.status.idle":"2024-04-05T17:49:56.078786Z","shell.execute_reply.started":"2024-04-05T17:49:52.299879Z","shell.execute_reply":"2024-04-05T17:49:56.076579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test=pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/3911565283.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:54:18.191830Z","iopub.execute_input":"2024-04-05T17:54:18.192336Z","iopub.status.idle":"2024-04-05T17:54:18.228140Z","shell.execute_reply.started":"2024-04-05T17:54:18.192298Z","shell.execute_reply":"2024-04-05T17:54:18.226831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def process_eeg_datatest(data_frame):\n#     \"\"\"\n#     Process EEG data.\n\n#     Parameters:\n#         data_frame (pd.DataFrame): EEG data in DataFrame format.\n\n#     Returns:\n#         mne.Epochs: MNE-Python Epochs object.\n#         np.ndarray: EEG data from the first epoch.\n#         np.ndarray: Power Spectral Density (PSD) estimates.\n#         np.ndarray: Corresponding frequencies.\n#     \"\"\"\n#     # Apply bandpass filter to EEG data\n#     fs = 250  # Sampling frequency\n#     lowcut = 1.0  # Lower cutoff frequency\n#     highcut = 50.0  # Upper cutoff frequency\n#     data_frame.dropna(inplace=True)\n#     data_frame_filtered = data_frame.apply(lambda x: butter_bandpass_filter(x, lowcut, highcut, fs), axis=0)\n    \n#     # Load EEG data into a Raw object\n#     data_frame_filtered_transposed = data_frame_filtered.T\n#     info = mne.create_info(ch_names=data_frame_filtered.columns.tolist(), sfreq=fs, ch_types='eeg')\n#     raw = mne.io.RawArray(data_frame_filtered_transposed.values, info)\n\n#     # Notch filter to remove power line interference (e.g., 50 Hz)\n#     raw.notch_filter(freqs=[50], picks='eeg')  # Adjust frequency as needed\n\n#     raw.drop_channels(['EKG'])\n\n#     montage = mne.channels.make_standard_montage('standard_1005')\n#     raw.set_montage(montage, on_missing='ignore')\n\n#     # Apply ICA for artifact removal\n#     ica = mne.preprocessing.ICA(n_components=len(data_frame.columns)-2, random_state=97, max_iter=800)\n#     ica.fit(raw)\n#     ica.exclude = [0, 1, 2, 3, 4, 5]  # Example muscle components\n#     ica.apply(raw)\n\n#     # Create epochs\n#     epochs = mne.make_fixed_length_epochs(raw, duration=5, overlap=1)\n\n#     # Optional preprocessing steps\n#     # 1. Artifact rejection\n#     epochs.drop_bad()\n    \n#     # Compute PSD estimates for each epoch using Welch's method\n#     psds = []\n#     freqs = None\n#     for epoch_data in epochs.get_data():\n#         f, p = welch(epoch_data.squeeze(), fs=fs, nperseg=256)\n#         if freqs is None:\n#             freqs = f\n#         psds.append(p)\n#     psds = np.array(psds)\n\n#     # 7. Interpolation of bad channels\n#     raw.interpolate_bads()\n\n#     # Visualize EEG data from the first epoch\n#     if len(epochs) > 0:\n#         eeg_data_epoch = epochs.get_data()\n#         # Optional: Plot EEG data from the first epoch\n#         times = epochs.times\n#         plt.figure(figsize=(10, 5))\n#         plt.plot(times, eeg_data_epoch[0].T)\n#         plt.title('EEG Data - First Epoch')\n#         plt.xlabel('Time (s)')\n#         plt.ylabel('EEG Amplitude (uV)')\n#         plt.show()\n\n#         # Optional: Plot drop log\n#         epochs.plot_drop_log()\n#     else:\n#         print(\"No remaining epochs after rejection\")\n\n#     return epochs, eeg_data_epoch, psds, freqs","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:55:44.295429Z","iopub.execute_input":"2024-04-05T17:55:44.295913Z","iopub.status.idle":"2024-04-05T17:55:44.312016Z","shell.execute_reply.started":"2024-04-05T17:55:44.295872Z","shell.execute_reply":"2024-04-05T17:55:44.309644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # assiginging preprocess data \n# epochstest, eeg_data_epochtest, psdstest, freqstest = process_eeg_datatest(test)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T17:55:49.725963Z","iopub.execute_input":"2024-04-05T17:55:49.726434Z","iopub.status.idle":"2024-04-05T17:56:01.949350Z","shell.execute_reply.started":"2024-04-05T17:55:49.726400Z","shell.execute_reply":"2024-04-05T17:56:01.946689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Apply wavelet denoising to the stacked EEG data\n# denoised_stacked_eeg_data_test = np.zeros_like(eeg_data_epochtest)\n# # Apply wavelet denoising to the stacked EEG data\n# #denoised_stacked_eeg_data_test = np.zeros_like(eeg_data_epoch_test)\n# for i in range(eeg_data_epochtest.shape[0]):  # Iterate over epoch\n#     for j in range(eeg_data_epochtest.shape[1]):  # Iterate over channel\n#         denoised_stacked_eeg_data_test[i, j,:] = denoise_signal(eeg_data_epochtest[i, j,:])\n\n# # Predict probabilities using the model\n# denoised_stacked_eeg_data_test = np.transpose(denoised_stacked_eeg_data_test, (0, 2, 1))\n# predicted_probabilities = model.predict(denoised_stacked_eeg_data_test)\n# #processed_epochs_test, eeg_data_epoch_test = process_eeg_datatest(test)\n# #featurestest=[]\n# #for d in eeg_data_epoch_test:\n#  #   featurestest.append(concatenate_features(d))\n# #features_arraytest=np.array(featurestest)\n# #y_predtest = classifier.predict(features_arraytest[:, :features_array.shape[1]])\n# #y_pred_probatest = classifier.predict_proba(features_arraytest[:, :features_array.shape[1]])\n\n# #y_predtest = model.predict(eeg_data_epoch_test)\n# #row_averages = np.mean(y_pred_probatest, axis=0)\n# average_probabilities = np.mean(predicted_probabilities, axis=0)\n# df=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\n# # Update the DataFrame with the predicted probabilities\n# df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']] = average_probabilities \n# # Save the updated DataFrame to a new CSV file\n# df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import matplotlib.pyplot as plt\n# from scipy.signal import hilbert, firwin, filtfilt\n# #from PyEMD import EMD\n\n# # Assuming stacked_eeg_data is a 3D array where the first dimension represents epochs,\n# # the second dimension represents channels, and the third dimension represents time samples.\n\n# # Define the sampling frequency and time vector\n# sampling_frequency = 200  # Assuming a sampling frequency of 100 Hz\n# time_vector = np.arange(stacked_eeg_data.shape[2]) / sampling_frequency\n\n# # Initialize EMD\n# #emd = EMD()\n\n# # Define filter parameters\n# filter_type = 'bandpass'\n# lowcut = 0.5  # Lower cutoff frequency in Hz\n# highcut = 50  # Upper cutoff frequency in Hz\n# filter_order = 100  # Filter order\n\n# # Design FIR filter\n# nyquist_freq = 0.5 * sampling_frequency\n# if filter_type == 'bandpass':\n#     fir_coeff = firwin(filter_order, [lowcut / nyquist_freq, highcut / nyquist_freq], pass_zero=False)\n# elif filter_type == 'lowpass':\n#     fir_coeff = firwin(filter_order, highcut / nyquist_freq, pass_zero=False)\n# elif filter_type == 'highpass':\n#     fir_coeff = firwin(filter_order, lowcut / nyquist_freq, pass_zero=False)\n\n# import numpy as np\n# import matplotlib.pyplot as plt\n# from scipy.signal import hilbert, firwin, filtfilt\n# from scipy.interpolate import InterpolatedUnivariateSpline\n\n# # Define functions for EMD\n# def find_extrema(signal):\n#     maxima = (signal[:-2] < signal[1:-1]) & (signal[1:-1] > signal[2:])\n#     minima = (signal[:-2] > signal[1:-1]) & (signal[1:-1] < signal[2:])\n#     return maxima, minima\n\n# def linear_interpolate(signal, indices):\n#     x = np.arange(len(signal))\n#     return np.interp(x, indices, signal[indices])\n\n# def extract_mean_envelope(signal):\n#     maxima, minima = find_extrema(signal)\n#     upper_env = linear_interpolate(signal, np.where(maxima)[0])\n#     lower_env = linear_interpolate(signal, np.where(minima)[0])\n#     return (upper_env + lower_env) / 2\n\n\n# def extract_imf(signal, tolerance=0.1, max_iterations=100):\n#     imf = signal.copy()\n#     for _ in range(max_iterations):\n#         mean_env = extract_mean_envelope(imf)\n#         imf_old = imf\n#         imf = imf - mean_env\n#         if np.sum(np.abs(imf_old - imf) / np.abs(imf)) < tolerance:\n#             return imf\n#     return imf\n\n# # Decompose the EEG signal for each epoch\n# imfs_all_epochstest = []\n# for epoch_data in eeg_data_epochtest:\n#     imfs_epochtest = []\n#     for channel_data in epoch_data:\n#         imfstest = emd(channel_data)\n#          # Apply Hilbert Transform to each IMF\n#         analytic_signalstest = [hilbert(imf) for imf in imfstest]\n        \n#         # Filter each IMF\n#         filtered_imfstest = []\n#         for analytic_signal in analytic_signalstest:\n#             filtered_signaltest = filtfilt(fir_coeff, 1.0, np.real(analytic_signal))  # Apply filtering\n#             filtered_imfstest.append(filtered_signaltest)\n#         imfs_epochtest.append(imfstest)\n#     imfs_all_epochstest.append(imfs_epochtest)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T18:15:05.014558Z","iopub.execute_input":"2024-04-05T18:15:05.015119Z","iopub.status.idle":"2024-04-05T18:15:27.380886Z","shell.execute_reply.started":"2024-04-05T18:15:05.015072Z","shell.execute_reply":"2024-04-05T18:15:27.379560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reconstructed_signaltest = np.zeros_like(eeg_data_epochtest[:])  # Initialize reconstructed signal\n\n# # Iterate through each epoch\n# for epoch_idx in range(len(imfs_all_epochstest)):\n#     # Iterate through each channel\n#     for channel_idx in range(len(imfs_all_epochs[epoch_idx])):\n#         # Initialize reconstructed signal for the current channel\n#         reconstructed_channel_signaltest = np.zeros_like(eeg_data_epochtest[epoch_idx][channel_idx])\n        \n#         # Iterate through each IMF\n#         for imf_idx in range(len(imfs_all_epochstest[epoch_idx][channel_idx])):\n#             # Add current IMF to the reconstructed channel signal\n#             reconstructed_channel_signaltest += imfs_all_epochstest[epoch_idx][channel_idx][imf_idx]\n        \n#         # Store the reconstructed signal for the current channel and epoch\n#         reconstructed_signaltest[epoch_idx][channel_idx] = reconstructed_channel_signaltest\n        ","metadata":{"execution":{"iopub.status.busy":"2024-04-05T18:16:23.975142Z","iopub.execute_input":"2024-04-05T18:16:23.975695Z","iopub.status.idle":"2024-04-05T18:16:23.991144Z","shell.execute_reply.started":"2024-04-05T18:16:23.975650Z","shell.execute_reply":"2024-04-05T18:16:23.989855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Predict probabilities using the model\n# denoised_stacked_eeg_data_test = np.transpose(reconstructed_signaltest, (0, 2, 1))\n# predicted_probabilities = model.predict(denoised_stacked_eeg_data_test)\n# average_probabilities = np.mean(predicted_probabilities, axis=0)\n# df=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\n# # Update the DataFrame with the predicted probabilities\n# df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']] = average_probabilities \n# # Save the updated DataFrame to a new CSV file\n# df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T18:16:33.932572Z","iopub.execute_input":"2024-04-05T18:16:33.933008Z","iopub.status.idle":"2024-04-05T18:16:34.112505Z","shell.execute_reply.started":"2024-04-05T18:16:33.932975Z","shell.execute_reply":"2024-04-05T18:16:34.111603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}