{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7465251,"sourceType":"datasetVersion","datasetId":4317718}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport math\n\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()\n\n# Discussion: Understanding Competition Data and EfficientNetB2 Starter - LB 0.43","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:19:33.105520Z","iopub.execute_input":"2024-03-14T09:19:33.105937Z","iopub.status.idle":"2024-03-14T09:19:34.815711Z","shell.execute_reply.started":"2024-03-14T09:19:33.105900Z","shell.execute_reply":"2024-03-14T09:19:34.814305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distribution of expert consensus\ndf['expert_consensus'].value_counts()[['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']].plot(kind='bar', title=\"Expert consensus distribution\")\nplt.tight_layout()\nplt.savefig('/kaggle/working/expertConsensusDistribution.png')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:19:34.818141Z","iopub.execute_input":"2024-03-14T09:19:34.818601Z","iopub.status.idle":"2024-03-14T09:19:35.386530Z","shell.execute_reply.started":"2024-03-14T09:19:34.818556Z","shell.execute_reply":"2024-03-14T09:19:35.385156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distrubution of average expert votes per class\nvotes = df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].apply(lambda x: x/x.sum(), axis=1)\n# print(votes[:10])\nvotesMean = votes.replace(0, np.NaN).mean(axis=0)\nvotesMean.plot(kind='bar', title='Average votes per class \\n(excluding cases of no (0) votes for each class)')\nplt.tight_layout()\nplt.savefig('/kaggle/working/averageVoteDistribution.png')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-14T09:39:24.737915Z","iopub.execute_input":"2024-03-14T09:39:24.738529Z","iopub.status.idle":"2024-03-14T09:39:45.845711Z","shell.execute_reply.started":"2024-03-14T09:39:24.738484Z","shell.execute_reply":"2024-03-14T09:39:45.844568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(votes[75:85])\nprint(votesMean)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:45:17.938049Z","iopub.execute_input":"2024-03-14T09:45:17.938580Z","iopub.status.idle":"2024-03-14T09:45:17.951644Z","shell.execute_reply.started":"2024-03-14T09:45:17.938544Z","shell.execute_reply":"2024-03-14T09:45:17.950518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking whether expert_consensus and votes change per EEG","metadata":{}},{"cell_type":"code","source":"#find out whether expert consensus change over the same eeg, different times\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n# consensus = pd.DataFrame(votes[['eeg_id', 'expert_consensus']].groupby('eeg_id')['expert_consensus'].nunique(), columns=['eeg_id', 'consensus_counts'])\nconsensus = df[['eeg_id', 'expert_consensus']].groupby('eeg_id')['expert_consensus'].nunique().reset_index()\nconsensus = consensus.rename(columns={'eeg_id': 'eeg_id', 'expert_consensus': 'expert_consensus_count'})\n# consensus = consensus.columns =['eeg_id', 'values_consensus']\n# print(type(consensus))\n# print(consensus)\nprint(consensus[consensus['expert_consensus_count'] != 1])\nprint(\"Number of eeg_id where the expert_consensus changes: \", len(consensus[consensus['expert_consensus_count'] != 1]))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:19:56.172319Z","iopub.execute_input":"2024-03-14T09:19:56.173638Z","iopub.status.idle":"2024-03-14T09:19:56.410153Z","shell.execute_reply.started":"2024-03-14T09:19:56.173595Z","shell.execute_reply":"2024-03-14T09:19:56.409004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find out whether vote distribution change over the same eeg, different times\nV = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nvotes = df[['eeg_id', 'eeg_label_offset_seconds', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\n\n# counts the number of votes per class per eeg_id\nvotes = votes.groupby('eeg_id')[V].nunique().reset_index()\n\n#if anywhere there were more than 1 different value in the votes, that means it changed\nvotes = votes.assign(changes = lambda x: x['seizure_vote'] + x['lpd_vote'] + x['gpd_vote'] + x['lrda_vote'] + x['grda_vote'] + x['other_vote'] > 6)\nvotesChange = votes[votes['changes']==True]\nvotesSame = votes[votes['changes']==False]\nprint(\"Number of eegs where the vote distribution changes at least once: \" + str(len(votesChange)))\nprint(\"Number of eegs where the vote distribution never changes: \" + str(len(votesSame)))\n\nvotesChangeMean = votesChange[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].mean(axis=0)\nvotesChangeMean.plot(kind='bar')\nplt.title('How many values are counted per vote per eeg \\n(i.e. higher means the vote switched more often per eeg)')\nplt.tight_layout()\nplt.savefig('/kaggle/working/changesVoteDistribution.png')\nplt.show()\n\n# Results:\n# there are 1807 eeg_ids where the distribution of votes changes\n# in the category \"other\" the votes change the most often\n# in the category \"grda\" and \"gdp\" the votes change the least often","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:19:56.411610Z","iopub.execute_input":"2024-03-14T09:19:56.412001Z","iopub.status.idle":"2024-03-14T09:19:57.049834Z","shell.execute_reply.started":"2024-03-14T09:19:56.411971Z","shell.execute_reply":"2024-03-14T09:19:57.048378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# V = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n# df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nvotesDevelopment = df[df['eeg_id'].isin(votesChange['eeg_id'])].copy()\n\n# calculate percentage of votes per class\nvotesDevelopment[V] = votesDevelopment[V].apply(lambda x: x/x.sum(), axis=1)\n\n# sort by offset per eeg_id so that we really only go forward in time\nvotesDevelopment = votesDevelopment.sort_values(by=['eeg_id', 'eeg_label_offset_seconds'])\n\n# group by eeg_id and then transform to see the differences\nvotesDevelopment[V] = votesDevelopment.groupby(['eeg_id'])[V].transform(lambda x: x.diff())\n\n# drop the NaN values which are always given to the first row of diff\nvotesDevelopment = votesDevelopment.dropna()\n\n# to only take the times into account when things change we remove cases of 0 (no change)\nvotesDevelopment = votesDevelopment[(votesDevelopment[V] != 0).any(axis=1)]\nvotesDevelopmentMean = votesDevelopment[V].mean(axis=0)\nvotesDevelopmentMean.plot(kind='bar')\nplt.title('Where do the votes move to (on average)?')\nplt.tight_layout()\nplt.savefig('/kaggle/working/developmentVoteDistribution.png')\nplt.show()\n\n# Results: Most of the times or the most extreme changes in the voting distribution changes occur\n# either because the class \"other\" loses votes \n# or because the class \"seizure\" gains votes","metadata":{"execution":{"iopub.status.busy":"2024-03-14T10:20:18.805123Z","iopub.execute_input":"2024-03-14T10:20:18.805625Z","iopub.status.idle":"2024-03-14T10:20:25.308013Z","shell.execute_reply.started":"2024-03-14T10:20:18.805588Z","shell.execute_reply":"2024-03-14T10:20:25.306855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using the notebook WaveNet Starter - [LB 0.52] to read raw EEG data\n***Everything below this until the next header is copied without any changes***","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt\n\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint( train.shape )\ndisplay( train.head() )\n\n# CHOICE TO CREATE OR LOAD EEGS FROM NOTEBOOK VERSION 1\nCREATE_EEGS = False\nTRAIN_MODEL = False","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:20:03.209977Z","iopub.execute_input":"2024-03-14T09:20:03.211082Z","iopub.status.idle":"2024-03-14T09:20:03.409314Z","shell.execute_reply.started":"2024-03-14T09:20:03.211045Z","shell.execute_reply":"2024-03-14T09:20:03.408018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet')\nFEATS = df.columns\nprint(f'There are {len(FEATS)} raw eeg features')\nprint( list(FEATS) )","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:20:03.410566Z","iopub.execute_input":"2024-03-14T09:20:03.410912Z","iopub.status.idle":"2024-03-14T09:20:03.598156Z","shell.execute_reply.started":"2024-03-14T09:20:03.410884Z","shell.execute_reply":"2024-03-14T09:20:03.597189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We will use the following subset of raw EEG features:')\nFEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\nFEAT2IDX = {x:y for x,y in zip(FEATS,range(len(FEATS)))}\nprint( list(FEATS) )","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:20:03.605554Z","iopub.execute_input":"2024-03-14T09:20:03.607983Z","iopub.status.idle":"2024-03-14T09:20:03.617785Z","shell.execute_reply.started":"2024-03-14T09:20:03.607945Z","shell.execute_reply":"2024-03-14T09:20:03.616638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eeg_from_parquet(parquet_path, display=False):\n    \n    # EXTRACT MIDDLE 50 SECONDS\n    eeg = pd.read_parquet(parquet_path, columns=FEATS)\n    rows = len(eeg)\n    offset = (rows-10_000)//2\n    eeg = eeg.iloc[offset:offset+10_000]\n    \n    if display: \n        plt.figure(figsize=(10,5))\n        offset = 0\n    \n    # CONVERT TO NUMPY\n    data = np.zeros((10_000,len(FEATS)))\n    for j,col in enumerate(FEATS):\n        \n        # FILL NAN\n        x = eeg[col].values.astype('float32')\n        m = np.nanmean(x)\n        if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n        else: x[:] = 0\n            \n        data[:,j] = x\n        \n        if display: \n            if j!=0: offset += x.max()\n            plt.plot(range(10_000),x-offset,label=col)\n            offset -= x.min()\n            \n    if display:\n        plt.legend()\n        name = parquet_path.split('/')[-1]\n        name = name.split('.')[0]\n        plt.title(f'EEG {name}',size=16)\n        plt.show()\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:20:03.619370Z","iopub.execute_input":"2024-03-14T09:20:03.620589Z","iopub.status.idle":"2024-03-14T09:20:03.634234Z","shell.execute_reply.started":"2024-03-14T09:20:03.620533Z","shell.execute_reply":"2024-03-14T09:20:03.633086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n\nall_eegs = {}\nDISPLAY = 4\nEEG_IDS = train.eeg_id.unique()\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='') \n    \n    # SAVE EEG TO PYTHON DICTIONARY OF NUMPY ARRAYS\n    data = eeg_from_parquet(f'{PATH}{eeg_id}.parquet', display=i<DISPLAY)              \n    all_eegs[eeg_id] = data\n    \n    if i==DISPLAY:\n        if CREATE_EEGS:\n            print(f'Processing {train.eeg_id.nunique()} eeg parquets... ',end='')\n        else:\n            print(f'Reading {len(EEG_IDS)} eeg NumPys from disk.')\n            break\n            \nif CREATE_EEGS: \n    np.save('eegs',all_eegs)\nelse:\n    all_eegs = np.load('/kaggle/input/brain-eegs/eegs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:20:03.636435Z","iopub.execute_input":"2024-03-14T09:20:03.637416Z","iopub.status.idle":"2024-03-14T09:21:49.768663Z","shell.execute_reply.started":"2024-03-14T09:20:03.637372Z","shell.execute_reply":"2024-03-14T09:21:49.767270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD TRAIN \ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\ntrain = df.groupby('eeg_id')[['patient_id']].agg('first')\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\n\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\ntrain = train.loc[train.eeg_id.isin(EEG_IDS)]\nprint('Train Data with unique eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:21:49.770614Z","iopub.execute_input":"2024-03-14T09:21:49.770969Z","iopub.status.idle":"2024-03-14T09:21:50.129194Z","shell.execute_reply.started":"2024-03-14T09:21:49.770940Z","shell.execute_reply":"2024-03-14T09:21:50.127992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import butter, lfilter\n\ndef butter_lowpass_filter(data, cutoff_freq=20, sampling_rate=200, order=4):\n    nyquist = 0.5 * sampling_rate\n    normal_cutoff = cutoff_freq / nyquist\n    b, a = butter(order, normal_cutoff, btype='low', analog=False)\n    filtered_data = lfilter(b, a, data, axis=0)\n    return filtered_data","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:21:50.130559Z","iopub.execute_input":"2024-03-14T09:21:50.130898Z","iopub.status.idle":"2024-03-14T09:21:50.784460Z","shell.execute_reply.started":"2024-03-14T09:21:50.130865Z","shell.execute_reply":"2024-03-14T09:21:50.783223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FREQS = [1,2,4,8,16][::-1]\nx = [all_eegs[EEG_IDS[0]][:,0]]\nfor k in FREQS:\n    x.append( butter_lowpass_filter(x[0], cutoff_freq=k) )\n\nplt.figure(figsize=(20,20))\nplt.plot(range(10_000),x[0], label='without filter')\nfor k in range(1,len(x)):\n    plt.plot(range(10_000),x[k]-k*(x[0].max()-x[0].min()), label=f'with filter {FREQS[k-1]}Hz')\nplt.legend()\nplt.title('Butter Low-Pass Filter Examples',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:21:50.785914Z","iopub.execute_input":"2024-03-14T09:21:50.787105Z","iopub.status.idle":"2024-03-14T09:21:51.746150Z","shell.execute_reply.started":"2024-03-14T09:21:50.787071Z","shell.execute_reply":"2024-03-14T09:21:51.744566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, batch_size=32, shuffle=False, eegs=all_eegs, mode='train',\n                 downsample=5): \n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.eegs = eegs\n        self.mode = mode\n        self.downsample = downsample\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int( np.ceil( len(self.data) / self.batch_size ) )\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X, y = self.__data_generation(indexes)\n        return X[:,::self.downsample,:], y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.data) )\n        if self.shuffle: np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' \n    \n        X = np.zeros((len(indexes),10_000,8),dtype='float32')\n        y = np.zeros((len(indexes),6),dtype='float32')\n        \n        sample = np.zeros((10_000,X.shape[-1]))\n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]      \n            data = self.eegs[row.eeg_id]\n            \n            # FEATURE ENGINEER\n            sample[:,0] = data[:,FEAT2IDX['Fp1']] - data[:,FEAT2IDX['T3']]\n            sample[:,1] = data[:,FEAT2IDX['T3']] - data[:,FEAT2IDX['O1']]\n            \n            sample[:,2] = data[:,FEAT2IDX['Fp1']] - data[:,FEAT2IDX['C3']]\n            sample[:,3] = data[:,FEAT2IDX['C3']] - data[:,FEAT2IDX['O1']]\n            \n            sample[:,4] = data[:,FEAT2IDX['Fp2']] - data[:,FEAT2IDX['C4']]\n            sample[:,5] = data[:,FEAT2IDX['C4']] - data[:,FEAT2IDX['O2']]\n            \n            sample[:,6] = data[:,FEAT2IDX['Fp2']] - data[:,FEAT2IDX['T4']]\n            sample[:,7] = data[:,FEAT2IDX['T4']] - data[:,FEAT2IDX['O2']]\n            \n            # STANDARDIZE\n            sample = np.clip(sample,-1024,1024)\n            sample = np.nan_to_num(sample, nan=0) / 32.0\n            \n            # BUTTER LOW-PASS FILTER\n            sample = butter_lowpass_filter(sample)\n            \n            X[j,] = sample\n            if self.mode!='test':\n                y[j] = row[TARGETS]\n            \n        return X,y","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:21:51.747965Z","iopub.execute_input":"2024-03-14T09:21:51.748420Z","iopub.status.idle":"2024-03-14T09:22:07.558519Z","shell.execute_reply.started":"2024-03-14T09:21:51.748379Z","shell.execute_reply":"2024-03-14T09:22:07.557444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Change here the batch size depending on how large a sample we want to investigate***","metadata":{}},{"cell_type":"code","source":"# gen = DataGenerator(train, shuffle=False)\ngen = DataGenerator(train, batch_size=600, shuffle=False)\n\nfor x,y in gen:\n    for k in range(4):\n        plt.figure(figsize=(20,4))\n        offset = 0\n        for j in range(x.shape[-1]):\n            if j!=0: offset -= x[k,:,j].min()\n            plt.plot(range(2_000),x[k,:,j]+offset,label=f'feature {j+1}')\n            offset += x[k,:,j].max()\n        tt = f'{y[k][0]:0.1f}'\n        for t in y[k][1:]:\n            tt += f', {t:0.1f}'\n        plt.title(f'EEG_Id = {EEG_IDS[k]}\\nTarget = {tt}',size=14)\n        plt.legend()\n        plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:07.560083Z","iopub.execute_input":"2024-03-14T09:22:07.560981Z","iopub.status.idle":"2024-03-14T09:22:11.687833Z","shell.execute_reply.started":"2024-03-14T09:22:07.560938Z","shell.execute_reply":"2024-03-14T09:22:11.686729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a dataframe out of the batch of training data we selected above","metadata":{}},{"cell_type":"code","source":"gen_s = gen\n# h = []\nallFrames = pd.DataFrame(columns =['eeg_id', 'feature 1', 'feature 2', 'feature 3', 'feature 4',\n       'feature 5', 'feature 6', 'feature 7', 'feature 8', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'])\nfor x,y in gen_s:\n#     h = h + [x, y]\n    for k in range(x.shape[0]-1):\n        offset = 0\n        frame = pd.DataFrame()\n        for j in range(x.shape[-1]):\n            if j!=0: offset -= x[k,:,j].min()\n            frame['eeg_id'] = EEG_IDS[k]\n            frame[f'feature {j+1}'] = [x[k,:,j]+offset,j+1][0]\n            offset += x[k,:,j].max()\n            frame[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']] = y[k]\n        tt = [y[k][0]]\n        for t in y[k][1:]:\n            tt = tt + t\n#         frame['votes'] = tt*(x.shape[-1])\n        allFrames = pd.concat([allFrames, frame])\n    break\n    \nprint(allFrames.columns)\nprint(len(allFrames.index))\nprint(allFrames.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:11.688976Z","iopub.execute_input":"2024-03-14T09:22:11.689918Z","iopub.status.idle":"2024-03-14T09:22:28.893394Z","shell.execute_reply.started":"2024-03-14T09:22:11.689885Z","shell.execute_reply":"2024-03-14T09:22:28.892211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate percentage of each vote\n#VOTES = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n# features[VOTES] = features[VOTES].apply(lambda x: x/x.sum(), axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:28.895058Z","iopub.execute_input":"2024-03-14T09:22:28.895522Z","iopub.status.idle":"2024-03-14T09:22:28.900532Z","shell.execute_reply.started":"2024-03-14T09:22:28.895489Z","shell.execute_reply":"2024-03-14T09:22:28.899223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration based on features","metadata":{}},{"cell_type":"code","source":"# Calculating features, using these sources:\n# https://www.hindawi.com/journals/cmmm/2015/576437/\n# The accumulations of all obtained features from all segments of all classes are employed as the input for the three different classifiers.\n# (X_mean) Mean, (X_med) Median, (X_mod) mode, (X_stand) standard deviation, (X_q1) first quartile, (X_q2) third quartile, (X_int) interquartile range, (X_skew) skewness, (X_kurt) kurtoses, (X_min) minimum , (X_max) and maximum\n\n# feature engineering: Notebook: Exploring EEG: A Beginner's Guide\n# statistical features (mean, standard deviation, maximum, and minimum)\n# \"The script iterates over rows in the train dataset, extracting EEG segments and applying the extract_frequency_band_features function to each channel in these segments. The extracted features from all channels are aggregated.\"\n\ndef calculate_features(values):\n    mean = np.mean(values)\n    median = np.median(values)\n    mode = st.mode(values)\n    std = np.std(values)\n    skw = skew(values)\n    kurt = kurtosis(values)\n    minimum = np.min(values)\n    maximum = np.max(values)\n    quant1 = np.quantile(values, 0.25)\n    quant3 = np.quantile(values, 0.75)\n    interquartile = iqr(values)\n    return {'mean': mean, 'median': median, 'mode': mode, 'std': std\n                       ,'first quantile': quant1, 'third quantile': quant3, 'interquartile range': interquartile\n                       , 'skew': skw, 'kurtosis': kurt, 'min': minimum, 'max': maximum}\n\ndef calculate_votes(votes_df):\n    seizure_vote = np.mean(votes_df['seizure_vote'].to_numpy())\n    lpd_vote = np.mean(votes_df['lpd_vote'].to_numpy())\n    gpd_vote = np.mean(votes_df['gpd_vote'].to_numpy())\n    lrda_vote = np.mean(votes_df['lrda_vote'].to_numpy())\n    grda_vote = np.mean(votes_df['grda_vote'].to_numpy())\n    other_vote = np.mean(votes_df['other_vote'].to_numpy())\n    return {'seizure_vote':seizure_vote, 'lpd_vote':lpd_vote, 'gpd_vote':gpd_vote\n                                       , 'lrda_vote':lrda_vote, 'grda_vote':grda_vote, 'other_vote':other_vote}","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:28.902278Z","iopub.execute_input":"2024-03-14T09:22:28.902680Z","iopub.status.idle":"2024-03-14T09:22:28.915376Z","shell.execute_reply.started":"2024-03-14T09:22:28.902636Z","shell.execute_reply":"2024-03-14T09:22:28.914068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:28.916901Z","iopub.execute_input":"2024-03-14T09:22:28.918291Z","iopub.status.idle":"2024-03-14T09:22:29.139970Z","shell.execute_reply.started":"2024-03-14T09:22:28.918255Z","shell.execute_reply":"2024-03-14T09:22:29.138858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats as st\nfrom scipy.stats import skew, kurtosis, iqr\nimport numpy as np\n\ndef features_of_sample(df_sample, functionsToUse):\n    #df_sample is the panda dataframe over wich the features should be calculated\n    #functionsToUse is the list of functions over which the features should be calculated\n    features = {}\n    targets = {}\n\n    for i in df_sample['eeg_id'].unique():\n        \n#         if len(train[train['eeg_id']==i]['expert_consensus'].unique()) > 1:\n#             break#TODO: for now I simply do this as I do not yet know how to deal with the offset to find the exact eeg with the exact label\n\n        eeg = df_sample[df_sample['eeg_id'] == i]\n#         label = train[train['eeg_id']==i]['expert_consensus'].unique()\n        for c in functionsToUse:\n            values = eeg[c].to_numpy()\n            features[i, c] = calculate_features(values)\n        targets[i] = calculate_votes(eeg[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']])\n#         for c in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n#             vote = eeg[c].to_numpy()\n#             targets[i, c] = np.mean(vote)\n\n\n    features = pd.DataFrame.from_dict(features).transpose()\n    features = features.astype(dtype= {'mean':'float64', 'median':'float64', 'std':'float64'\n                                       , 'first quantile':'float64', 'third quantile':'float64', 'interquartile range':'float64'\n                                       , 'skew':'float64', 'kurtosis':'float64', 'min':'float64', 'max':'float64'})\n    targets = pd.DataFrame.from_dict(targets).transpose()\n    targets = targets.astype(dtype = {'seizure_vote':'float64', 'lpd_vote':'float64', 'gpd_vote':'float64'\n                                       , 'lrda_vote':'float64', 'grda_vote':'float64', 'other_vote':'float64'})\n    return features, targets\n\nfeatures, targets = features_of_sample(allFrames, ['feature 1', 'feature 2', 'feature 3', 'feature 4','feature 5', 'feature 6', 'feature 7', 'feature 8'])\n# print(allFrames.columns)\nprint(features.head())\nprint(targets.head())\n# print(features[1628180742])\n# print(len(features['labels']))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:22:29.141178Z","iopub.execute_input":"2024-03-14T09:22:29.141512Z","iopub.status.idle":"2024-03-14T09:24:51.732410Z","shell.execute_reply.started":"2024-03-14T09:22:29.141483Z","shell.execute_reply":"2024-03-14T09:24:51.730078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate percentage of each vote\n# MEAN_VOTES = ['mean_seizure_vote', 'mean_lpd_vote', 'mean_gpd_vote', 'mean_lrda_vote', 'mean_grda_vote', 'mean_other_vote']\n# features[MEAN_VOTES] = features[MEAN_VOTES].apply(lambda x: x/x.sum(), axis=1)\nprint(features.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:24:51.735058Z","iopub.execute_input":"2024-03-14T09:24:51.735632Z","iopub.status.idle":"2024-03-14T09:24:51.744142Z","shell.execute_reply.started":"2024-03-14T09:24:51.735589Z","shell.execute_reply":"2024-03-14T09:24:51.742705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # calculating the features over the groups LEFT TEMPORAL (LT), LEFT PARASAGITTAL (LP), RIGHT PARASAGITTAL  (RP), RIGHT TEMPORAL (RT) CHAIN\n# averageChains = allFrames\n# # ['eeg_id', 'feature 1', 'feature 2', 'feature 3', 'feature 4','feature 5', 'feature 6', 'feature 7', 'feature 8', 'seizure_vote','lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n# averageChains['LT'] = averageChains[['feature 1', 'feature 2']].mean(axis=1)\n# averageChains['LP'] = averageChains[['feature 3', 'feature 4']].mean(axis=1)\n# averageChains['RP'] = averageChains[['feature 5', 'feature 6']].mean(axis=1)\n# averageChains['RT'] = averageChains[['feature 7', 'feature 8']].mean(axis=1)\nfeatures.loc[1626798710]","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:24:51.745890Z","iopub.execute_input":"2024-03-14T09:24:51.746396Z","iopub.status.idle":"2024-03-14T09:24:51.779072Z","shell.execute_reply.started":"2024-03-14T09:24:51.746335Z","shell.execute_reply":"2024-03-14T09:24:51.777555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating euclidean distance within and between groups\n#currently leaving out mode as a feature because it is no float, would need to change that first from the tuple\nfrom scipy.spatial import distance_matrix\nimport seaborn as sns\nimport sys\n\n\nTHRESHOLD = 0.9\nVOTES = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nFEATS = ['mean', 'median', 'mode', 'std', 'first quantile', 'third quantile','interquartile range', 'skew', 'kurtosis', 'min', 'max']\n#for each expert vote, we calculate distance to each expert vote and get the weighted average (weighted by percentage of votes)\ndef avg_distances(df_features_of_eegs, list_features_to_compare):\n    avg_dist = {}\n    numberCases = {}\n    for label1 in VOTES:\n        cases1 = targets[targets[label1] > THRESHOLD].index\n        numberCases[label1] = len(cases1)\n        cases1 = features.loc[cases1]\n        cases1 = cases1[[list_features_to_compare]]\n    \n        if not cases1.empty:\n            for label2 in VOTES:\n                cases2 = targets[targets[label2] > THRESHOLD].index\n                cases2 = features.loc[cases2]\n                cases2 = cases2[[list_features_to_compare]]\n\n                if not cases2.empty:\n                    dists = distance_matrix(cases1.values, cases2.values)\n                    avg_dist[label1, label2] = (sum(sum(dists)))/((len(cases1)/8)*(len(cases2)/8))\n    return avg_dist, numberCases\n\navg_dist, numberCases = avg_distances(features, 'median')\nprint(avg_dist)\n# print(numberCases)\n# print(casesTargets2)\n# # print(cases1.head())\n# print(len(cases1))\n# print(cases1.index[1190:])\n# print(len(cases1.index))\n# print(c1.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:24:51.780859Z","iopub.execute_input":"2024-03-14T09:24:51.781198Z","iopub.status.idle":"2024-03-14T09:24:52.387033Z","shell.execute_reply.started":"2024-03-14T09:24:51.781170Z","shell.execute_reply":"2024-03-14T09:24:52.385632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig, ax = plt.subplots(10)\ndef plot_each_feature(df_f):\n    for i, f in enumerate(['mean', 'median', 'std', 'first quantile', 'third quantile','interquartile range', 'skew', 'kurtosis', 'min', 'max']):\n#         print(f)\n        avg_dist, numberCases = avg_distances(df_f, f)\n        dists = pd.DataFrame.from_dict([avg_dist]).transpose()\n        dists.index = pd.MultiIndex.from_tuples(dists.index)\n        dists = dists.stack().unstack(level=1)\n        dists = dists.reset_index( level = [1] )\n        dists = dists.loc[:, dists.columns != 'level_1']\n        dists = dists[['gpd_vote','grda_vote','lpd_vote', 'lrda_vote', 'seizure_vote']]\n        dists = dists.iloc[[0, 1, 2, 3, 5]]\n    #     print(dists.head())\n    #     print(\"Average distances between points of groups -- smaller means more similar\")\n        sns.heatmap(dists, annot=True)\n        plt.title(str(f) + ' : '+ str(numberCases))\n        plt.tight_layout()\n        plt.savefig('/kaggle/working/{}.png'.format(f))\n        plt.show()\n        \nplot_each_feature(features)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:24:52.389446Z","iopub.execute_input":"2024-03-14T09:24:52.390084Z","iopub.status.idle":"2024-03-14T09:25:01.254365Z","shell.execute_reply.started":"2024-03-14T09:24:52.390043Z","shell.execute_reply":"2024-03-14T09:25:01.252960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # calculating the features over the groups LEFT TEMPORAL (LT), LEFT PARASAGITTAL (LP), RIGHT PARASAGITTAL  (RP), RIGHT TEMPORAL (RT) CHAIN\nchains = allFrames\n# ['eeg_id', 'feature 1', 'feature 2', 'feature 3', 'feature 4','feature 5', 'feature 6', 'feature 7', 'feature 8', 'seizure_vote','lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\nchains['LT'] = chains[['feature 1', 'feature 2']].mean(axis=1)\nchains['LP'] = chains[['feature 3', 'feature 4']].mean(axis=1)\nchains['RP'] = chains[['feature 5', 'feature 6']].mean(axis=1)\nchains['RT'] = chains[['feature 7', 'feature 8']].mean(axis=1)\n\naverageChains = chains[['eeg_id', 'LT', 'LP', 'RP', 'RT', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote','other_vote']]\nfeatures, targets = features_of_sample(averageChains, ['LT', 'LP', 'RP', 'RT'])\n# avg_dist, numberCases = avg_distances(features, 'median')\nplot_each_feature(features)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:25:01.256442Z","iopub.execute_input":"2024-03-14T09:25:01.257201Z","iopub.status.idle":"2024-03-14T09:27:23.371598Z","shell.execute_reply.started":"2024-03-14T09:25:01.257168Z","shell.execute_reply":"2024-03-14T09:27:23.370411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VOTES = ['gpd_vote','grda_vote','lpd_vote', 'lrda_vote', 'seizure_vote']\ndef similarity_sin(df_features_eeg, list_features_compare):\n    sin_curve = np.sin(np.linspace(-666*np.pi, 667*np.pi, 2000))\n    sinFeats = calculate_features(sin_curve)\n    sinFeats = pd.DataFrame.from_dict([sinFeats])\n    sinFeats = sinFeats[[list_features_compare]]\n    # makes it repeat 8 times, because we compare all of the 8 measures of the EEG to it\n    sinFeats8 = pd.DataFrame(np.repeat(sinFeats.values, 8, axis=0))\n    sinFeats8.columns = sinFeats.columns\n    \n    for label in VOTES:\n        cases = targets[targets[label] > THRESHOLD].index\n        cases = features.loc[cases]\n        cases = cases[[list_features_compare]]\n\n        if not cases.empty:\n#             print(type(cases)) \n            print(cases.head())\n#             print(type(sinFeats8))\n            print(sinFeats8.head())\n            dists = distance_matrix(cases.values, sinFeats8.values)\n            avg_dist[label, \"sinus\"] = (sum(sum(dists)))/((len(cases)/8))\n            \n    return avg_dist","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:27:23.377356Z","iopub.execute_input":"2024-03-14T09:27:23.377745Z","iopub.status.idle":"2024-03-14T09:27:23.388751Z","shell.execute_reply.started":"2024-03-14T09:27:23.377712Z","shell.execute_reply":"2024-03-14T09:27:23.387463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a = np.sin(np.linspace(-666*np.pi, 667*np.pi, 2000))\n# print(a[:10])\nsin_curve = np.sin(np.linspace(-666*np.pi, 667*np.pi, 2000))\nsinFeats = calculate_features(sin_curve)\nsinFeats = pd.DataFrame.from_dict([sinFeats])\nsinFeats8 = pd.DataFrame(np.repeat(sinFeats.values, 8, axis=0))\nsinFeats8.columns = sinFeats.columns\n# print(sinFeats8)\nFEATS = ['mean', 'median', 'std', 'first quantile', 'third quantile','interquartile range', 'skew', 'kurtosis', 'min', 'max']\nfor i, f in enumerate(FEATS):\n#     print(f)\n    dist = similarity_sin(features, f)\n    dists = pd.DataFrame.from_dict([dist]).transpose()\n#     dists.index = pd.MultiIndex.from_tuples(dists.index)\n#     dists = dists.stack().unstack(level=1)\n#     dists = dists.reset_index( level = [1] )\n#     dists = dists.loc[:, dists.columns != 'level_1']\n#     dists = dists[['gpd_vote','grda_vote','lpd_vote', 'lrda_vote', 'seizure_vote']]\n#     dists = dists.iloc[[0, 1, 2, 3, 5]]\n    sns.heatmap(dists, annot=True)\n    plt.title(str(f) + ' : '+ str(numberCases))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:27:23.390187Z","iopub.execute_input":"2024-03-14T09:27:23.390800Z","iopub.status.idle":"2024-03-14T09:27:30.058933Z","shell.execute_reply.started":"2024-03-14T09:27:23.390767Z","shell.execute_reply":"2024-03-14T09:27:30.057562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: make our own spectograms","metadata":{"execution":{"iopub.status.busy":"2024-03-14T09:27:30.061447Z","iopub.execute_input":"2024-03-14T09:27:30.061893Z","iopub.status.idle":"2024-03-14T09:27:30.067161Z","shell.execute_reply.started":"2024-03-14T09:27:30.061854Z","shell.execute_reply":"2024-03-14T09:27:30.065999Z"},"trusted":true},"execution_count":null,"outputs":[]}]}