{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if '.csv' in filename:\n            print(os.path.join(dirname, filename))\n            \n            \nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-22T08:07:20.765973Z","iopub.execute_input":"2024-01-22T08:07:20.766328Z","iopub.status.idle":"2024-01-22T08:07:41.967576Z","shell.execute_reply.started":"2024-01-22T08:07:20.766298Z","shell.execute_reply":"2024-01-22T08:07:41.966627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-22T08:07:41.968937Z","iopub.execute_input":"2024-01-22T08:07:41.970119Z","iopub.status.idle":"2024-01-22T08:07:42.275177Z","shell.execute_reply.started":"2024-01-22T08:07:41.970081Z","shell.execute_reply":"2024-01-22T08:07:42.274141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corr_eeg(spec_id):\n    \n    eeg_ids = df_train[df_train.spectrogram_id == spec_id]['eeg_id'].unique()\n#      eeg_ids is list of eeg-id for this specid\n    \n    for eeg_id in eeg_ids: \n        eeg_id_str = str(eeg_id) + '.parquet'        \n    \n        # plot spectogram \n        df = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/' + eeg_id_str )\n        \n        # Assuming you have a DataFrame named 'df' and a correlation limit\n        correlation_limit = 0.8\n\n        # Calculate the correlation matrix\n        correlation_matrix = df.corr()\n\n        # Iterate through the columns and find pairs with correlations above the limit\n        correlation_pairs = []\n        for i in range(len(correlation_matrix.columns)):\n            for j in range(i):\n                if abs(correlation_matrix.iloc[i, j]) > correlation_limit:\n                    pair = (correlation_matrix.columns[i], correlation_matrix.columns[j])\n                    correlation_pairs.append(pair)\n\n        # Print the column pairs with correlations above the limit\n        for pair in correlation_pairs:\n            print(f\"Columns {pair[0]} and {pair[1]} have a correlation of {correlation_matrix.loc[pair[0], pair[1]]}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-22T08:22:41.790601Z","iopub.execute_input":"2024-01-22T08:22:41.791010Z","iopub.status.idle":"2024-01-22T08:22:41.799836Z","shell.execute_reply.started":"2024-01-22T08:22:41.790979Z","shell.execute_reply":"2024-01-22T08:22:41.798602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = 'Seizure' # 'GPD' 'LRDA' 'GRDA' 'LPD'\nnp.random.seed(42)\nspec_ids = df_train[df_train.expert_consensus == target]['spectrogram_id'].unique()\nran_spec_ids = np.random.choice(spec_ids, size=3, replace=False)\nran_spec_ids # this is a list of random spec-ids \n\nfor spec_id in ran_spec_ids:\n    print('###############################################################################################################')\n    corr_eeg(spec_id)\n    print('###############################################################################################################')\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-22T08:23:29.794069Z","iopub.execute_input":"2024-01-22T08:23:29.794428Z","iopub.status.idle":"2024-01-22T08:23:29.955923Z","shell.execute_reply.started":"2024-01-22T08:23:29.794398Z","shell.execute_reply":"2024-01-22T08:23:29.954769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for target -1 which is seizure, make a list of columns pairs which have correlation > 0.8 \n# keep size = 3 to get a better solution . \n# repeat this above step by changing the target to other 4 targets one by one. 'GPD' 'LRDA' 'GRDA' 'LPD'\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_spec(spec_id): \n    spec_id =  str(spec_id) + '.parquet' # '1000646093.parquet'\n    # eeg_id_1 = \n\n    # plot spectogram \n    spec_file = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/' + spec_id )\n\n    # Display the spectrogram using imshow\n    plt.imshow(np.log(spec_file.T)) # , cmap='viridis', origin='lower')  # Adjust colormap as desired\n    plt.colorbar(label='Spectral Power')  # Add a colorbar for interpretation\n    plt.xlabel('Time')\n    plt.ylabel('Frequency')\n    #     plt.title('Spectrogram')\n    plt.title(str(spec_id))\n    plt.show()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-21T17:25:19.758554Z","iopub.execute_input":"2024-01-21T17:25:19.758872Z","iopub.status.idle":"2024-01-21T17:25:19.764485Z","shell.execute_reply.started":"2024-01-21T17:25:19.758819Z","shell.execute_reply":"2024-01-21T17:25:19.763553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_eeg_ids_from_spect_id(spec_id, sig_to_plot, plot_no):\n    eeg_ids = df_train[df_train.spectrogram_id == spec_id]['eeg_id'].unique()\n    \n    for eeg_id in eeg_ids: \n        eeg_id_str = str(eeg_id) + '.parquet'        \n    \n        # plot spectogram \n        df = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/' + eeg_id_str )\n        \n#         print(df.head(5))\n\n#             df.columns\n        plt.figure()\n        if plot_no == 1:\n            plt.plot(df.index[0:200], df[sig_to_plot][0:200])\n            plt.title('spectrum id' + str(spec_id) + '  ' + 'eeg_id' + ' ' +  str(eeg_id) )\n        elif plot_no == 2:\n            plt.plot(df.index[0:10000:50], df[sig_to_plot][0:10000:50])\n            plt.title('spectrum id' + str(spec_id) + '  ' + 'eeg_id' + str(eeg_id) )\n        else:\n            plt.plot(df.index, df[sig_to_plot]) # , label=signal)  \n            plt.title('spectrum id' + str(spec_id) + '  ' + 'eeg_id' + str(eeg_id) )\n    \n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-21T17:25:19.766432Z","iopub.execute_input":"2024-01-21T17:25:19.766637Z","iopub.status.idle":"2024-01-21T17:25:19.836659Z","shell.execute_reply.started":"2024-01-21T17:25:19.766619Z","shell.execute_reply":"2024-01-21T17:25:19.835837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom scipy.signal import butter, filtfilt","metadata":{"execution":{"iopub.status.busy":"2024-01-21T17:25:22.027350Z","iopub.execute_input":"2024-01-21T17:25:22.028604Z","iopub.status.idle":"2024-01-21T17:25:22.076948Z","shell.execute_reply.started":"2024-01-21T17:25:22.028568Z","shell.execute_reply":"2024-01-21T17:25:22.076084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Filtering - Band Pass","metadata":{}},{"cell_type":"code","source":"# EEG Plots # Change the target and the plot no from 1 to 2 to 3 for different length and sampling of data, refer to function plot_eeg_ids_from_spect_id for more details\nimport numpy as np \ntarget = 'Seizure' # 'GPD' 'LRDA' 'GRDA' 'LPD'\nnp.random.seed(42)\nspec_ids = df_train[df_train.expert_consensus == target]['spectrogram_id'].unique()\nran_spec_ids = np.random.choice(spec_ids, size=1, replace=False)\nran_spec_ids\n\nfor spec_id in ran_spec_ids:\n    eeg_ids = df_train[df_train.spectrogram_id == spec_id]['eeg_id'].unique()\n    \n    for eeg_id in eeg_ids: \n        eeg_id_str = str(eeg_id) + '.parquet'        \n    \n        # plot spectogram \n        df = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/' + eeg_id_str )\n\n\n        \n\n        # Define the sampling rate (in Hz) of your EEG signal\n        # You need to know this beforehand; it's a fixed value based on how the EEG data was collected.\n        fs = 200  # For example, 256 Hz\n\n        # Define the frequency bands in Hz\n        alpha_band = (8, 13)\n        beta_band = (13, 30)\n        gamma_band = (30, 80)\n\n        # Create a bandpass filter for each frequency band\n        def bandpass_filter(data, lowcut, highcut, fs, order=5):\n            nyquist = 0.5 * fs\n            low = lowcut / nyquist\n            high = highcut / nyquist\n            b, a = butter(order, [low, high], btype='band')\n            y = filtfilt(b, a, data)\n            return y\n\n        # Apply the bandpass filter to each EEG signal column\n        eeg_columns = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz',\n                       'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2']\n\n        # Assuming 'df' is your DataFrame with EEG signal data\n        filtered_signals = {}\n        for column in eeg_columns:\n            filtered_signals[f'{column}_alpha'] = bandpass_filter(df[column], alpha_band[0], alpha_band[1], fs)\n            filtered_signals[f'{column}_beta'] = bandpass_filter(df[column], beta_band[0], beta_band[1], fs)\n            filtered_signals[f'{column}_gamma'] = bandpass_filter(df[column], gamma_band[0], gamma_band[1], fs)\n\n        # Convert the filtered signals dictionary to a DataFrame if needed\n        filtered_df = pd.DataFrame(filtered_signals)\n\n\n        sig_to_plot = 'Fp1'    \n#         sig_to_plot_filt = sig_to_plot + '_alpha'\n        sig_to_plot_filt = sig_to_plot + '_gamma'\n        plt.figure()\n        plt.plot(df[sig_to_plot][0:200])\n        plt.plot(filtered_signals[sig_to_plot_filt][0:200])\n#         plt.plot(df['Fp1'])\n#         plt.plot(filtered_signals['Fp1_alpha'])\n#         plt.plot(filtered_signals['Fp1_beta'])\n#         plt.plot(filtered_signals['Fp1_gamma'])\n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-21T15:33:23.387301Z","iopub.execute_input":"2024-01-21T15:33:23.387705Z","iopub.status.idle":"2024-01-21T15:33:23.695087Z","shell.execute_reply.started":"2024-01-21T15:33:23.387674Z","shell.execute_reply":"2024-01-21T15:33:23.693552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot the EEG Signals tasked to you, for every Target class and observe if visually you can see any similarity in the signal that can be associated to a particular target class","metadata":{}},{"cell_type":"code","source":"# EEG Plots # Change the target and the plot no from 1 to 2 to 3 for different length and sampling of data, refer to function plot_eeg_ids_from_spect_id for more details\nimport numpy as np \ntarget = 'Seizure' # 'GPD' 'LRDA' 'GRDA' 'LPD'\nnp.random.seed(42)\nspec_ids = df_train[df_train.expert_consensus == target]['spectrogram_id'].unique()\nran_spec_ids = np.random.choice(spec_ids, size=5, replace=False)\nran_spec_ids\n\nfor spec_id in ran_spec_ids:\n    plot_eeg_ids_from_spect_id(spec_id, 'Fp1', 1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-21T17:27:04.526966Z","iopub.execute_input":"2024-01-21T17:27:04.527258Z","iopub.status.idle":"2024-01-21T17:27:05.952473Z","shell.execute_reply.started":"2024-01-21T17:27:04.527235Z","shell.execute_reply":"2024-01-21T17:27:05.951586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectrogram Analysis","metadata":{}},{"cell_type":"code","source":"# randomly select 5 spectrogram for class \n\ntarget = 'Seizure' # 'LRDA' 'GRDA' 'LPD' 'Seizure'\n# plot the 5 spectograms \nimport numpy as np \nnp.random.seed(42)\nspec_ids = df_train[df_train.expert_consensus =='Seizure']['spectrogram_id'].unique()\nran_spec_ids = np.random.choice(spec_ids, size=5, replace=False)\nran_spec_ids\n\nfor spec_id in ran_spec_ids:\n    plot_spec(spec_id)","metadata":{"execution":{"iopub.status.busy":"2024-01-21T17:29:18.876796Z","iopub.execute_input":"2024-01-21T17:29:18.877131Z","iopub.status.idle":"2024-01-21T17:29:20.689858Z","shell.execute_reply.started":"2024-01-21T17:29:18.877107Z","shell.execute_reply":"2024-01-21T17:29:20.688870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}