{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import MinMaxScaler\n\nimport pywt\nfrom scipy.signal import butter, sosfilt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nfrom tqdm.notebook import tqdm, trange\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-24T06:26:52.915382Z","iopub.execute_input":"2024-01-24T06:26:52.916766Z","iopub.status.idle":"2024-01-24T06:26:55.559429Z","shell.execute_reply.started":"2024-01-24T06:26:52.916720Z","shell.execute_reply":"2024-01-24T06:26:55.558137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_eegs = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*')\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:27:16.250105Z","iopub.execute_input":"2024-01-24T06:27:16.250537Z","iopub.status.idle":"2024-01-24T06:27:16.959772Z","shell.execute_reply.started":"2024-01-24T06:27:16.250502Z","shell.execute_reply":"2024-01-24T06:27:16.958655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sec2idx(time, sr=200):\n    # this fuction get time point in sec and return corresponding row id in eeg file.\n    return time*200\n\n# custom dataloader\nclass EEGDataset:\n    def __init__(self, base_path, train):\n        self.base_path = base_path\n        self.eegs = glob.glob(base_path+'*')\n        self.train = train\n        \n    def __len__(self):\n        return len(self.train)\n    \n    def __getitem__(self, idx):\n        row = self.train.iloc[idx]\n        file = self.base_path+str(row['eeg_id'])+'.parquet'\n        st_idx = int(sec2idx(row['eeg_label_offset_seconds']))\n        end_idx = int(st_idx+sec2idx(50))\n        #print(row)\n        \n        eeg = pd.read_parquet(file)\n        X = eeg.iloc[st_idx:end_idx]\n        y = row['expert_consensus']\n        \n        return X, y\n    \ndataset = EEGDataset('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/', train)    ","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:32:24.474875Z","iopub.execute_input":"2024-01-24T06:32:24.475362Z","iopub.status.idle":"2024-01-24T06:32:24.554855Z","shell.execute_reply.started":"2024-01-24T06:32:24.475330Z","shell.execute_reply":"2024-01-24T06:32:24.553628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize Some Examples","metadata":{}},{"cell_type":"code","source":"START = 10\n\nfor i in range(5):\n    eeg, cat = dataset[START+i]\n\n    eeg.plot(subplots=True, figsize=(10, 7))\n    plt.title(cat)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:35:20.753482Z","iopub.execute_input":"2024-01-24T06:35:20.753921Z","iopub.status.idle":"2024-01-24T06:35:48.205082Z","shell.execute_reply.started":"2024-01-24T06:35:20.753884Z","shell.execute_reply":"2024-01-24T06:35:48.203618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot correlation of eeg channels with each other \nIDX = 0\n\neeg,_ = dataset[IDX]\nplt.figure(figsize=(20, 20))\nsns.heatmap(eeg.corr(), annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:34:28.258077Z","iopub.execute_input":"2024-01-24T06:34:28.258515Z","iopub.status.idle":"2024-01-24T06:34:29.860704Z","shell.execute_reply.started":"2024-01-24T06:34:28.258479Z","shell.execute_reply":"2024-01-24T06:34:29.859442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Fuctions for Band Extraction","metadata":{}},{"cell_type":"code","source":"def eeg_bandpass(eeg, min_freq, max_freq):\n    #this function get signal and range of frequencies we interested and filter out every other frequency components\n    # from signal\n    \n    sos_lp = butter(10, max_freq, 'lp', fs=200, output='sos')\n    sos_hp = butter(10, min_freq, 'hp', fs=200, output='sos')\n    \n    eeg_low = sosfilt(sos_lp, eeg)\n    eeg_high = sosfilt(sos_hp, eeg_low)\n    \n    return eeg_high\n\ndef eeg2band(eeg, band='alpha'):\n    bands = {'alpha':(8, 12), 'beta':(12,30), 'gamma':(35,99), 'delta':(0.5,4), 'theta':(4,8)}\n    band_range = bands[band]\n    min_freq, max_freq = band_range[0], band_range[1]\n    \n    return eeg_bandpass(eeg, min_freq, max_freq)","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:38:05.229311Z","iopub.execute_input":"2024-01-24T06:38:05.229800Z","iopub.status.idle":"2024-01-24T06:38:05.239476Z","shell.execute_reply.started":"2024-01-24T06:38:05.229757Z","shell.execute_reply":"2024-01-24T06:38:05.238321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract and Plotting Bands","metadata":{}},{"cell_type":"code","source":"START = 10\n\n# appling our function on first channel in selected eeg files\nfor i in range(5):\n    eeg,cat = dataset[START+i]\n    eeg_channel_0 = eeg.values[:,0]\n    \n    plt.subplots(5, 1, figsize=(15, 5))\n    for j, band in enumerate(['delta','theta' ,'alpha', 'beta', 'gamma']):\n        filtered = eeg2band(eeg_channel_0, band)\n        \n        plt.subplot(5,1,j+1)\n        #plt.figure(figsize=(15, 1))\n        plt.plot(eeg_channel_0)\n        plt.plot(filtered)\n        plt.title(band)\n    plt.suptitle(cat)        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-24T06:44:22.150272Z","iopub.execute_input":"2024-01-24T06:44:22.150820Z","iopub.status.idle":"2024-01-24T06:44:27.355435Z","shell.execute_reply.started":"2024-01-24T06:44:22.150777Z","shell.execute_reply":"2024-01-24T06:44:27.354256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}