{"metadata":{"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7570342,"sourceType":"datasetVersion","datasetId":4407194},{"sourceId":7776446,"sourceType":"datasetVersion","datasetId":4550181}],"dockerImageVersionId":30637,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":101.739864,"end_time":"2024-03-03T17:15:15.283704","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-03-03T17:13:33.54384","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Loading Meta Data","metadata":{}},{"cell_type":"code","source":"import os, random, pandas as pd, numpy as np\nfrom scipy.signal import butter, lfilter\nimport matplotlib.pyplot as plt\n\nTARGETS = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nMETA = ['spectrogram_id','spectrogram_label_offset_seconds','patient_id','expert_consensus']\ntrain = train.groupby('eeg_id')[META+TARGETS\n                           ].agg({**{m:'first' for m in META},**{t:'sum' for t in TARGETS}}).reset_index() \ntrain[TARGETS] = train[TARGETS]/train[TARGETS].values.sum(axis=1,keepdims=True)\ntrain.columns = ['eeg_id','spec_id','offset','patient_id','target'] + TARGETS\nprint(train.head(1).to_string())","metadata":{"papermill":{"duration":0.024968,"end_time":"2024-03-03T17:13:52.310258","exception":false,"start_time":"2024-03-03T17:13:52.28529","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-09-26T02:23:31.640993Z","iopub.execute_input":"2024-09-26T02:23:31.641230Z","iopub.status.idle":"2024-09-26T02:23:35.223097Z","shell.execute_reply.started":"2024-09-26T02:23:31.641205Z","shell.execute_reply":"2024-09-26T02:23:35.222410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading 8-channel EEG signal","metadata":{}},{"cell_type":"code","source":"%%time\nall_raw_eegs = np.load('/kaggle/input/hms-eeg/eegs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T02:23:35.224340Z","iopub.execute_input":"2024-09-26T02:23:35.224604Z","iopub.status.idle":"2024-09-26T02:24:49.730134Z","shell.execute_reply.started":"2024-09-26T02:23:35.224578Z","shell.execute_reply":"2024-09-26T02:24:49.729407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## DATA GENERATOR","metadata":{"papermill":{"duration":0.007968,"end_time":"2024-03-03T17:13:52.37087","exception":false,"start_time":"2024-03-03T17:13:52.362902","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FEATS2 = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\nFEAT2IDX = {x:y for x,y in zip(FEATS2,range(len(FEATS2)))}\n    \nclass DataGenerator():\n    'Generates data for Keras'\n    def __init__(self, data, raw_eegs=None): \n        self.data = data\n        self.raw_eegs = raw_eegs\n        \n    def __len__(self):\n        return self.data.shape[0]\n\n    def __getitem__(self, index):\n        X, y = self.data_generation(index)\n        return X, y\n    \n    def __call__(self):\n        for i in range(self.__len__()):\n            yield self.__getitem__(i)\n                \n    def data_generation(self, index):\n        X,y = self.generate_raw(index)\n        return X,y\n    \n    def generate_raw(self,index):\n        X = np.zeros((10_000,8),dtype='float32')\n        y = np.zeros((6,),dtype='float32')\n        \n        row = self.data.iloc[index]\n        eeg = self.raw_eegs[row.eeg_id]\n            \n        # FEATURE ENGINEER\n        X[:,0] = eeg[:,FEAT2IDX['Fp1']] - eeg[:,FEAT2IDX['T3']]\n        X[:,1] = eeg[:,FEAT2IDX['T3']] - eeg[:,FEAT2IDX['O1']]\n            \n        X[:,2] = eeg[:,FEAT2IDX['Fp1']] - eeg[:,FEAT2IDX['C3']]\n        X[:,3] = eeg[:,FEAT2IDX['C3']] - eeg[:,FEAT2IDX['O1']]\n            \n        X[:,4] = eeg[:,FEAT2IDX['Fp2']] - eeg[:,FEAT2IDX['C4']]\n        X[:,5] = eeg[:,FEAT2IDX['C4']] - eeg[:,FEAT2IDX['O2']]\n            \n        X[:,6] = eeg[:,FEAT2IDX['Fp2']] - eeg[:,FEAT2IDX['T4']]\n        X[:,7] = eeg[:,FEAT2IDX['T4']] - eeg[:,FEAT2IDX['O2']]\n            \n        # STANDARDIZE\n        X = np.clip(X,-1024,1024)\n        X = np.nan_to_num(X, nan=0) / 32.0\n            \n        # BUTTER LOW-PASS FILTER\n        X = self.butter_lowpass_filter(X)\n        # Downsample\n        X = X[::5,:]\n        \n        y[:] = row[TARGETS]\n        self.raw_eegs[row.eeg_id] = X\n        return X,y\n        \n    def butter_lowpass_filter(self, data, cutoff_freq=20, sampling_rate=200, order=4):\n        nyquist = 0.5 * sampling_rate\n        normal_cutoff = cutoff_freq / nyquist\n        b, a = butter(order, normal_cutoff, btype='low', analog=False)\n        filtered_data = lfilter(b, a, data, axis=0)\n        return filtered_data","metadata":{"papermill":{"duration":2.248907,"end_time":"2024-03-03T17:13:54.628134","exception":false,"start_time":"2024-03-03T17:13:52.379227","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-09-26T02:24:49.731151Z","iopub.execute_input":"2024-09-26T02:24:49.731403Z","iopub.status.idle":"2024-09-26T02:24:49.743070Z","shell.execute_reply.started":"2024-09-26T02:24:49.731373Z","shell.execute_reply":"2024-09-26T02:24:49.742444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save Processed EEG","metadata":{}},{"cell_type":"code","source":"gen = DataGenerator(train,raw_eegs=all_raw_eegs)\nfor i, (x,y) in enumerate(gen):\n    if i%1000==0: print(i,', ',end='')\n    pass\n\nnp.save('eegs_processed.npy',gen.raw_eegs)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T02:24:49.744387Z","iopub.execute_input":"2024-09-26T02:24:49.744645Z","iopub.status.idle":"2024-09-26T02:25:47.530809Z","shell.execute_reply.started":"2024-09-26T02:24:49.744619Z","shell.execute_reply":"2024-09-26T02:25:47.529981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display EEG signal","metadata":{}},{"cell_type":"code","source":"x = gen.raw_eegs[568657]\nplt.figure(figsize=(20,4))\noffset = 0\nfor j in range(x.shape[-1]):\n    if j!=0: offset -= x[:,j].min()\n    plt.plot(range(2_000),x[:,j]+offset,label=f'feature {j+1}')\n    offset += x[:,j].max()\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T02:25:47.531718Z","iopub.execute_input":"2024-09-26T02:25:47.531967Z","iopub.status.idle":"2024-09-26T02:25:47.908596Z","shell.execute_reply.started":"2024-09-26T02:25:47.531942Z","shell.execute_reply":"2024-09-26T02:25:47.907747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}