{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7465251,"sourceType":"datasetVersion","datasetId":4317718},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Feature Based Classification using RAW EEG Features!\n\nBased on [this notebook][https://www.kaggle.com/code/cdeotte/wavenet-starter-lb-0-52]\n\nThis model only uses two features. We can engineer more features and/or modify the model architecture to improve CV score and LB score. Furthermore we can build 1 model which inputs both spectrogram images and eeg waveforms. The two EEG features in this notebook are:\n* feature 1 : `Fp1 minus O1`\n* feature 2 : `Fp2 minus O2`\n\nFeature 1 is the beginning of the montage chains `LL` and `LP` minus the ending of montage `LL` and `LP`. And feature 2 is the beginning of the montage chains `RL` and `RP` minus the ending of montage `RL` and `RP`.","metadata":{}},{"cell_type":"markdown","source":"# Load Train Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt\n\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntrain = train[train['expert_consensus'] != 'Other']\n\nprint( train.shape )\ndisplay( train.head() )\n\n# CHOICE TO CREATE OR LOAD EEGS FROM NOTEBOOK VERSION 1\nCREATE_EEGS = False\nTRAIN_MODEL = False","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:52:09.015020Z","iopub.execute_input":"2024-03-12T14:52:09.015473Z","iopub.status.idle":"2024-03-12T14:52:10.560118Z","shell.execute_reply.started":"2024-03-12T14:52:09.015426Z","shell.execute_reply":"2024-03-12T14:52:10.559162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Raw EEG Features","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet')\nFEATS = df.columns\nprint(f'There are {len(FEATS)} raw eeg features')\nprint( list(FEATS) )","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:52:10.561750Z","iopub.execute_input":"2024-03-12T14:52:10.562557Z","iopub.status.idle":"2024-03-12T14:52:10.726640Z","shell.execute_reply.started":"2024-03-12T14:52:10.562519Z","shell.execute_reply":"2024-03-12T14:52:10.725427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We will use the following subset of raw EEG features:')\nFEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\nFEAT2IDX = {x:y for x,y in zip(FEATS,range(len(FEATS)))}\nprint( list(FEATS) )","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:52:10.730587Z","iopub.execute_input":"2024-03-12T14:52:10.730981Z","iopub.status.idle":"2024-03-12T14:52:10.738530Z","shell.execute_reply.started":"2024-03-12T14:52:10.730945Z","shell.execute_reply":"2024-03-12T14:52:10.737327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eeg_from_parquet(parquet_path, display=False):\n    \n    # EXTRACT MIDDLE 50 SECONDS\n    eeg = pd.read_parquet(parquet_path, columns=FEATS)\n    rows = len(eeg)\n    offset = (rows-10_000)//2\n    eeg = eeg.iloc[offset:offset+10_000]\n    \n    if display: \n        plt.figure(figsize=(10,5))\n        offset = 0\n    \n    # CONVERT TO NUMPY\n    data = np.zeros((10_000,len(FEATS)))\n    for j,col in enumerate(FEATS):\n        \n        # FILL NAN\n        x = eeg[col].values.astype('float32')\n        m = np.nanmean(x)\n        if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n        else: x[:] = 0\n            \n        data[:,j] = x\n        \n        if display: \n            if j!=0: offset += x.max()\n            plt.plot(range(10_000),x-offset,label=col)\n            offset -= x.min()\n            \n    if display:\n        plt.legend()\n        name = parquet_path.split('/')[-1]\n        name = name.split('.')[0]\n        plt.title(f'EEG {name}',size=16)\n        plt.show()\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:52:10.741676Z","iopub.execute_input":"2024-03-12T14:52:10.742632Z","iopub.status.idle":"2024-03-12T14:52:10.754535Z","shell.execute_reply.started":"2024-03-12T14:52:10.742580Z","shell.execute_reply":"2024-03-12T14:52:10.753582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nall_eegs = {}\nDISPLAY = 0\nSAMPLE = 400\nEEG_IDS = train.eeg_id.unique()\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='') \n    \n    # SAVE EEG TO PYTHON DICTIONARY OF NUMPY ARRAYS\n    data = eeg_from_parquet(f'{PATH}{eeg_id}.parquet', display=i<DISPLAY)              \n    all_eegs[eeg_id] = data\n    \n    if i==DISPLAY:\n        if CREATE_EEGS:\n            print(f'Processing {train.eeg_id.nunique()} eeg parquets... ',end='')\n        else:\n            print(f'Reading {len(EEG_IDS)} eeg NumPys from disk.')\n            break\n            \nif CREATE_EEGS: \n    np.save('eegs',all_eegs)\nelse:\n    all_eegs = np.load('/kaggle/input/brain-eegs/eegs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:52:10.756190Z","iopub.execute_input":"2024-03-12T14:52:10.756534Z","iopub.status.idle":"2024-03-12T14:53:43.338274Z","shell.execute_reply.started":"2024-03-12T14:52:10.756503Z","shell.execute_reply":"2024-03-12T14:53:43.336913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deduplicate Train EEG Id","metadata":{}},{"cell_type":"code","source":"# LOAD TRAIN \ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\ntrain = df.groupby('eeg_id')[['patient_id']].agg('first')\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\ntrain = train.loc[train.eeg_id.isin(EEG_IDS)]\nprint('Train Data with unique eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:43.339805Z","iopub.execute_input":"2024-03-12T14:53:43.340191Z","iopub.status.idle":"2024-03-12T14:53:43.627302Z","shell.execute_reply.started":"2024-03-12T14:53:43.340128Z","shell.execute_reply":"2024-03-12T14:53:43.626151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Butter Low-Pass Filter","metadata":{}},{"cell_type":"code","source":"from scipy.signal import butter, lfilter\n\ndef butter_lowpass_filter(data, cutoff_freq=20, sampling_rate=200, order=4):\n    nyquist = 0.5 * sampling_rate\n    normal_cutoff = cutoff_freq / nyquist\n    b, a = butter(order, normal_cutoff, btype='low', analog=False)\n    filtered_data = lfilter(b, a, data, axis=0)\n    return filtered_data","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:43.628814Z","iopub.execute_input":"2024-03-12T14:53:43.630024Z","iopub.status.idle":"2024-03-12T14:53:44.732305Z","shell.execute_reply.started":"2024-03-12T14:53:43.629981Z","shell.execute_reply":"2024-03-12T14:53:44.731359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FREQS = [1,2,4,8,16][::-1]\nx = [all_eegs[EEG_IDS[0]][:,0]]\nfor k in FREQS:\n    x.append( butter_lowpass_filter(x[0], cutoff_freq=k) )\n\nplt.figure(figsize=(20,20))\nplt.plot(range(10_000),x[0], label='without filter')\nfor k in range(1,len(x)):\n    plt.plot(range(10_000),x[k]-k*(x[0].max()-x[0].min()), label=f'with filter {FREQS[k-1]}Hz')\nplt.legend()\nplt.title('Butter Low-Pass Filter Examples',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:44.733703Z","iopub.execute_input":"2024-03-12T14:53:44.734078Z","iopub.status.idle":"2024-03-12T14:53:45.769822Z","shell.execute_reply.started":"2024-03-12T14:53:44.734043Z","shell.execute_reply":"2024-03-12T14:53:45.768919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader with Butter Low-Pass Filter","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, batch_size=SAMPLE, shuffle=False, eegs=all_eegs, mode='train',\n                 downsample=5): \n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.eegs = eegs\n        self.mode = mode\n        self.downsample = downsample\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int( np.ceil( len(self.data) / self.batch_size ) )\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X, y = self.__data_generation(indexes)\n        return X[:,::self.downsample,:], y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.data) )\n        if self.shuffle: np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' \n    \n        X = np.zeros((len(indexes),10_000,8),dtype='float32')\n        y = np.zeros((len(indexes),6),dtype='float32')\n        \n        sample = np.zeros((10_000,X.shape[-1]))\n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]      \n            data = self.eegs[row.eeg_id]\n            \n            # FEATURE ENGINEER\n            sample[:,0] = data[:,FEAT2IDX['Fp1']] - data[:,FEAT2IDX['T3']]\n            sample[:,1] = data[:,FEAT2IDX['T3']] - data[:,FEAT2IDX['O1']]\n            \n            sample[:,2] = data[:,FEAT2IDX['Fp1']] - data[:,FEAT2IDX['C3']]\n            sample[:,3] = data[:,FEAT2IDX['C3']] - data[:,FEAT2IDX['O1']]\n            \n            sample[:,4] = data[:,FEAT2IDX['Fp2']] - data[:,FEAT2IDX['C4']]\n            sample[:,5] = data[:,FEAT2IDX['C4']] - data[:,FEAT2IDX['O2']]\n            \n            sample[:,6] = data[:,FEAT2IDX['Fp2']] - data[:,FEAT2IDX['T4']]\n            sample[:,7] = data[:,FEAT2IDX['T4']] - data[:,FEAT2IDX['O2']]\n            \n            # STANDARDIZE\n            sample = np.clip(sample,-1024,1024)\n            sample = np.nan_to_num(sample, nan=0) / 32.0\n            \n            # BUTTER LOW-PASS FILTER\n            sample = butter_lowpass_filter(sample)\n            \n            X[j,] = sample\n            if self.mode!='test':\n                y[j] = row[TARGETS]\n            \n        return X,y","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:45.771021Z","iopub.execute_input":"2024-03-12T14:53:45.771524Z","iopub.status.idle":"2024-03-12T14:53:54.662759Z","shell.execute_reply.started":"2024-03-12T14:53:45.771492Z","shell.execute_reply":"2024-03-12T14:53:54.661704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display Data Loader","metadata":{}},{"cell_type":"code","source":"gen = DataGenerator(train, shuffle=False)\n\nfor x,y in gen:\n    for k in range(4):\n        plt.figure(figsize=(20,4))\n        offset = 0\n        for j in range(x.shape[-1]):\n            if j!=0: offset -= x[k,:,j].min()\n            plt.plot(range(2_000),x[k,:,j]+offset,label=f'feature {j+1}')\n            offset += x[k,:,j].max()\n        tt = f'{y[k][0]:0.1f}'\n        for t in y[k][1:]:\n            tt += f', {t:0.1f}'\n        plt.title(f'EEG_Id = {EEG_IDS[k]}\\nTarget = {tt}',size=14)\n        plt.legend()\n        plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:54.668494Z","iopub.execute_input":"2024-03-12T14:53:54.669293Z","iopub.status.idle":"2024-03-12T14:53:58.187222Z","shell.execute_reply.started":"2024-03-12T14:53:54.669242Z","shell.execute_reply":"2024-03-12T14:53:58.185894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initialize GPUs","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport tensorflow as tf\nprint('TensorFlow version =',tf.__version__)\n\n# USE MULTIPLE GPUS\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:58.189580Z","iopub.execute_input":"2024-03-12T14:53:58.190009Z","iopub.status.idle":"2024-03-12T14:53:58.221001Z","shell.execute_reply.started":"2024-03-12T14:53:58.189970Z","shell.execute_reply":"2024-03-12T14:53:58.219645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE MIXED PRECISION\nMIX = True\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:58.222868Z","iopub.execute_input":"2024-03-12T14:53:58.223360Z","iopub.status.idle":"2024-03-12T14:53:58.234312Z","shell.execute_reply.started":"2024-03-12T14:53:58.223311Z","shell.execute_reply":"2024-03-12T14:53:58.233439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Features","metadata":{}},{"cell_type":"code","source":"# Make dataframe of the eegs\n\nallFrames = pd.DataFrame(columns =['eeg_id', 'feature 1', 'feature 2', 'feature 3', 'feature 4',\n       'feature 5', 'feature 6', 'feature 7', 'feature 8', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'])\nfor x,y in gen:\n#     h = h + [x, y]\n    for k in range(x.shape[0]-1):\n        offset = 0\n        frame = pd.DataFrame()\n        for j in range(x.shape[-1]):\n            if j!=0: offset -= x[k,:,j].min()\n            frame['eeg_id'] = EEG_IDS[k]\n            frame[f'feature {j+1}'] = [x[k,:,j],j+1][0]\n            frame[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']] = y[k]\n        tt = [y[k][0]]\n        for t in y[k][1:]:\n            tt = tt + t\n        allFrames = pd.concat([allFrames, frame])\n    break\n    \nprint(allFrames.columns)\nprint(len(allFrames.index))\nprint(allFrames.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:53:58.236279Z","iopub.execute_input":"2024-03-12T14:53:58.237199Z","iopub.status.idle":"2024-03-12T14:54:07.994149Z","shell.execute_reply.started":"2024-03-12T14:53:58.237136Z","shell.execute_reply":"2024-03-12T14:54:07.992957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distrubution of average expert votes per class\nvotes = allFrames[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].apply(lambda x: x/x.sum(), axis=1)\n# print(votes[:10])\nvotesMean = votes.mean(axis=0)\nvotesMean.plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:54:07.995514Z","iopub.execute_input":"2024-03-12T14:54:07.995901Z","iopub.status.idle":"2024-03-12T14:57:24.493858Z","shell.execute_reply.started":"2024-03-12T14:54:07.995868Z","shell.execute_reply":"2024-03-12T14:57:24.492472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating the average over the features to get the chains \n# LEFT TEMPORAL (LT), LEFT PARASAGITTAL (LP), RIGHT PARASAGITTAL  (RP), RIGHT TEMPORAL (RT) CHAIN\nchains = allFrames\n\nchains['LT'] = chains[['feature 1', 'feature 2']].mean(axis=1)\nchains['LP'] = chains[['feature 3', 'feature 4']].mean(axis=1)\nchains['RP'] = chains[['feature 5', 'feature 6']].mean(axis=1)\nchains['RT'] = chains[['feature 7', 'feature 8']].mean(axis=1)\n\naverageChains = chains[['eeg_id', 'LT', 'LP', 'RP', 'RT', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\nprint(averageChains.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:57:24.495564Z","iopub.execute_input":"2024-03-12T14:57:24.496155Z","iopub.status.idle":"2024-03-12T14:57:24.979588Z","shell.execute_reply.started":"2024-03-12T14:57:24.496113Z","shell.execute_reply":"2024-03-12T14:57:24.978309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_features(values):\n    mean = np.mean(values)\n    median = np.median(values)\n    mode = st.mode(values)\n    std = np.std(values)\n    skw = skew(values)\n    kurt = kurtosis(values)\n    minimum = np.min(values)\n    maximum = np.max(values)\n    quant1 = np.quantile(values, 0.25)\n    quant3 = np.quantile(values, 0.75)\n    interquartile = iqr(values)\n    return {'mean': mean, 'median': median, 'mode': mode, 'std': std\n                       ,'first quantile': quant1, 'third quantile': quant3, 'interquartile range': interquartile\n                       , 'skew': skw, 'kurtosis': kurt, 'min': minimum, 'max': maximum}\n\ndef calculate_votes(votes_df):\n    seizure_vote = np.mean(votes_df['seizure_vote'].to_numpy())\n    lpd_vote = np.mean(votes_df['lpd_vote'].to_numpy())\n    gpd_vote = np.mean(votes_df['gpd_vote'].to_numpy())\n    lrda_vote = np.mean(votes_df['lrda_vote'].to_numpy())\n    grda_vote = np.mean(votes_df['grda_vote'].to_numpy())\n    other_vote = np.mean(votes_df['other_vote'].to_numpy())\n    return {'seizure_vote':seizure_vote, 'lpd_vote':lpd_vote, 'gpd_vote':gpd_vote\n                                       , 'lrda_vote':lrda_vote, 'grda_vote':grda_vote, 'other_vote':other_vote}","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:57:24.981051Z","iopub.execute_input":"2024-03-12T14:57:24.981425Z","iopub.status.idle":"2024-03-12T14:57:24.992100Z","shell.execute_reply.started":"2024-03-12T14:57:24.981391Z","shell.execute_reply":"2024-03-12T14:57:24.990692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats as st\nfrom scipy.stats import skew, kurtosis, iqr\nimport numpy as np\n\n# calculate the features of an EEG per chain\ndef features_of_sample(df_sample):\n    features = {}\n    targets = {}\n    \n    #this just calculates features over the entire eeg and does not take any offset into account\n    #therefore it also just calculates the average over the votes\n    for i in df_sample['eeg_id'].unique():\n        \n        eeg = df_sample[df_sample['eeg_id'] == i]\n        for c in ['LT',  'LP', 'RT', 'RP']:\n            values = eeg[c].to_numpy()\n            features[i, c] = calculate_features(values)\n        targets[i] = calculate_votes(eeg[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']])\n\n\n    features = pd.DataFrame.from_dict(features).transpose()\n    features = features.astype(dtype= {'mean':'float64', 'median':'float64', 'std':'float64'\n                                       , 'first quantile':'float64', 'third quantile':'float64', 'interquartile range':'float64'\n                                       , 'skew':'float64', 'kurtosis':'float64', 'min':'float64', 'max':'float64'})\n    targets = pd.DataFrame.from_dict(targets).transpose()\n    targets = targets.astype(dtype = {'seizure_vote':'float64', 'lpd_vote':'float64', 'gpd_vote':'float64'\n                                       , 'lrda_vote':'float64', 'grda_vote':'float64', 'other_vote':'float64'})\n    return features, targets\n\n# features, targets = features_of_sample(allFrames)\nfeatures, targets = features_of_sample(averageChains)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:57:24.993800Z","iopub.execute_input":"2024-03-12T14:57:24.994324Z","iopub.status.idle":"2024-03-12T14:57:54.063525Z","shell.execute_reply.started":"2024-03-12T14:57:24.994276Z","shell.execute_reply":"2024-03-12T14:57:54.062380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build X and y as training set and validation set","metadata":{}},{"cell_type":"code","source":"# function to find the majority vote\ndef majority_vote(values):\n    return np.argmax(values)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T14:57:54.064936Z","iopub.execute_input":"2024-03-12T14:57:54.065291Z","iopub.status.idle":"2024-03-12T14:57:54.070718Z","shell.execute_reply.started":"2024-03-12T14:57:54.065258Z","shell.execute_reply":"2024-03-12T14:57:54.069737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nimport math\n\n#choosing features to use\nfeatures = features.loc[:, features.columns != 'mode']\nfeatures = features.loc[:, features.columns != 'mean']\nfeatures = features.loc[:, features.columns != 'median']\n# features = features.loc[:, features.columns != 'std']\nfeatures = features.loc[:, features.columns != 'first quantile']\nfeatures = features.loc[:, features.columns != 'third quantile']\n# features = features.loc[:, features.columns != 'interquartile range']\n# features = features.loc[:, features.columns != 'skew']\n# features = features.loc[:, features.columns != 'kurtosis']\nfeatures = features.loc[:, features.columns != 'min']\nfeatures = features.loc[:, features.columns != 'max']\nx = features.groupby(level=0, sort=False).apply(lambda x: x.values)\nx = x.apply(lambda x: x.ravel()).values#.ravel()\nx = np.stack(x)\ny = targets.values\ny_class = np.apply_along_axis(majority_vote, 1, y)\n\n#leave out the \"other\" votes\n\nsize = len(y)\ntrain_size = math.floor(size/5*4)\nval_size = math.floor(size/5)\nprint(\"overall sample size: \", size)\nprint(\"training sample size: \", train_size)\nprint(\"validation sample size: \",val_size)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:09.116822Z","iopub.execute_input":"2024-03-12T15:25:09.117528Z","iopub.status.idle":"2024-03-12T15:25:09.223699Z","shell.execute_reply.started":"2024-03-12T15:25:09.117465Z","shell.execute_reply":"2024-03-12T15:25:09.221691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Python implemented way of selecting train and validation samples\nfrom sklearn.model_selection import train_test_split\n# from sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n#standardize feature data\ntrans = StandardScaler()\nx = trans.fit_transform(x)\n\nX_train, X_val, y_train, y_val = train_test_split(x, y_class, test_size = 0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:09.226453Z","iopub.execute_input":"2024-03-12T15:25:09.227069Z","iopub.status.idle":"2024-03-12T15:25:09.241473Z","shell.execute_reply.started":"2024-03-12T15:25:09.227011Z","shell.execute_reply":"2024-03-12T15:25:09.239493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Model (Support Vector Machine)","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics\nfrom sklearn import svm\n\n#Create a svm Classifier\nclf = svm.SVC(decision_function_shape='ovo')\n\n#Train the model using the training sets\n# clf.fit(X_train, y_class_train)\nclf.fit(X_train, y_train)\n\n#Predict the response for test dataset\ny_pred_class = clf.predict(X_val)\n\n# Model Accuracy: how often is the classifier correct?\nprint(\"Accuracy:\",metrics.accuracy_score(y_val, y_pred_class))","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:09.243930Z","iopub.execute_input":"2024-03-12T15:25:09.244450Z","iopub.status.idle":"2024-03-12T15:25:09.274912Z","shell.execute_reply.started":"2024-03-12T15:25:09.244380Z","shell.execute_reply":"2024-03-12T15:25:09.272562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nvotes = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nfig, ax = plt.subplots()\n\ncnf_matrix = metrics.confusion_matrix(y_val, y_pred_class)\n\nsns.heatmap(pd.DataFrame(cnf_matrix), annot=True, xticklabels = votes, yticklabels = votes)\nax.xaxis.set_label_position(\"top\")\nplt.tight_layout()\nplt.title('Confusion matrix of the SVM')\nplt.ylabel('Actual label')\nplt.xlabel('Predicted label')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:09.279078Z","iopub.execute_input":"2024-03-12T15:25:09.279623Z","iopub.status.idle":"2024-03-12T15:25:10.368465Z","shell.execute_reply.started":"2024-03-12T15:25:09.279569Z","shell.execute_reply":"2024-03-12T15:25:10.364702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Logistic Regression Model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nclf = LogisticRegression(random_state=0).fit(X_train, y_train)\ny_pred_class = clf.predict(X_val)\ny_pred_prob =  clf.predict_proba(X_val)\n\nprint(\"Accuracy:\",metrics.accuracy_score(y_val, y_pred_class))","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:10.372182Z","iopub.execute_input":"2024-03-12T15:25:10.374412Z","iopub.status.idle":"2024-03-12T15:25:10.419372Z","shell.execute_reply.started":"2024-03-12T15:25:10.374338Z","shell.execute_reply":"2024-03-12T15:25:10.417239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"votes =['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nfig, ax = plt.subplots()\n\n\ncnf_matrix = metrics.confusion_matrix(y_val, y_pred_class)\n\nsns.heatmap(pd.DataFrame(cnf_matrix), annot=True, xticklabels = votes, yticklabels = votes)\n# ax.xaxis.set_label_position(\"top\")\nplt.tight_layout()\nplt.title('Confusion matrix of the LR')\nplt.ylabel('Actual label')\nplt.xlabel('Predicted label')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:10.421474Z","iopub.execute_input":"2024-03-12T15:25:10.421955Z","iopub.status.idle":"2024-03-12T15:25:11.353581Z","shell.execute_reply.started":"2024-03-12T15:25:10.421913Z","shell.execute_reply":"2024-03-12T15:25:11.351492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit to Kaggle LB","metadata":{}},{"cell_type":"code","source":"# del all_eegs, train; gc.collect()\n# test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\n# print('Test shape:',test.shape)\n# test.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:11.356439Z","iopub.execute_input":"2024-03-12T15:25:11.358355Z","iopub.status.idle":"2024-03-12T15:25:11.367977Z","shell.execute_reply.started":"2024-03-12T15:25:11.358261Z","shell.execute_reply":"2024-03-12T15:25:11.365114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all_eegs2 = {}\n# DISPLAY = 1\n# EEG_IDS2 = test.eeg_id.unique()\n# PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\n\n# print('Processing Test EEG parquets...'); print()\n# for i,eeg_id in enumerate(EEG_IDS2):\n        \n#     # SAVE EEG TO PYTHON DICTIONARY OF NUMPY ARRAYS\n#     data = eeg_from_parquet(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n#     all_eegs2[eeg_id] = data","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:11.371481Z","iopub.execute_input":"2024-03-12T15:25:11.372432Z","iopub.status.idle":"2024-03-12T15:25:11.382872Z","shell.execute_reply.started":"2024-03-12T15:25:11.372328Z","shell.execute_reply":"2024-03-12T15:25:11.380482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # INFER MLP ON TEST\n# preds = []\n# model = build_model()\n# test_gen = DataGenerator(test, shuffle=False, batch_size=64, eegs=all_eegs2, mode='test')\n\n# print('Inferring test... ',end='')\n# for i in range(FOLDS_TO_TRAIN):\n#     print(f'fold {i+1}, ',end='')\n#     if TRAIN_MODEL:\n#         model.load_weights(f'WaveNet_Model/WaveNet_fold{i}.h5')\n#     else:\n#         model.load_weights(f'/kaggle/input/brain-eegs/WaveNet_Model/WaveNet_fold{i}.h5')\n#     pred = model.predict(test_gen, verbose=0)\n#     preds.append(pred)\n# pred = np.mean(preds,axis=0)\n# print()\n# print('Test preds shape',pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:11.387606Z","iopub.execute_input":"2024-03-12T15:25:11.388280Z","iopub.status.idle":"2024-03-12T15:25:11.398436Z","shell.execute_reply.started":"2024-03-12T15:25:11.388212Z","shell.execute_reply":"2024-03-12T15:25:11.395934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # CREATE SUBMISSION.CSV\n# from IPython.display import display\n\n# sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\n# sub[TARGETS] = pred\n# sub.to_csv('submission.csv',index=False)\n# print('Submission shape',sub.shape)\n# display( sub.head() )\n\n# # SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\n# print('Sub row 0 sums to:',sub.iloc[0,-6:].sum())","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:25:11.400324Z","iopub.execute_input":"2024-03-12T15:25:11.400919Z","iopub.status.idle":"2024-03-12T15:25:11.416497Z","shell.execute_reply.started":"2024-03-12T15:25:11.400878Z","shell.execute_reply":"2024-03-12T15:25:11.414502Z"},"trusted":true},"execution_count":null,"outputs":[]}]}