{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re\nfrom string import punctuation\nfrom nltk.tokenize import word_tokenize\nfrom keras.utils import to_categorical\nimport string\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split\nimport spacy\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report\nfrom sklearn.metrics import log_loss\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.over_sampling import SMOTE, SMOTENC\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.combine import SMOTEENN\nfrom collections import Counter\nimport tqdm\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport joblib\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n#         print(os.path.join(dirname, filename))\n\npd.set_option('display.max_rows', 200)\npd.set_option('display.max_columns', 150)\npd.set_option('display.width', 1000)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-21T07:07:21.712833Z","iopub.execute_input":"2024-06-21T07:07:21.714280Z","iopub.status.idle":"2024-06-21T07:07:58.505839Z","shell.execute_reply.started":"2024-06-21T07:07:21.714204Z","shell.execute_reply":"2024-06-21T07:07:58.504757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EEG - HMS","metadata":{}},{"cell_type":"code","source":"Base_Path = '/kaggle/input/hms-harmful-brain-activity-classification'\ndf_train = pd.read_csv(f'{Base_Path}/train.csv')\ndf_test = pd.read_csv(f'{Base_Path}/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:07:58.507730Z","iopub.execute_input":"2024-06-21T07:07:58.508063Z","iopub.status.idle":"2024-06-21T07:07:58.677751Z","shell.execute_reply.started":"2024-06-21T07:07:58.508035Z","shell.execute_reply":"2024-06-21T07:07:58.676725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:07:58.679057Z","iopub.execute_input":"2024-06-21T07:07:58.679454Z","iopub.status.idle":"2024-06-21T07:07:58.695143Z","shell.execute_reply.started":"2024-06-21T07:07:58.679419Z","shell.execute_reply":"2024-06-21T07:07:58.694011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:07:58.697752Z","iopub.execute_input":"2024-06-21T07:07:58.698156Z","iopub.status.idle":"2024-06-21T07:07:58.713097Z","shell.execute_reply.started":"2024-06-21T07:07:58.698119Z","shell.execute_reply":"2024-06-21T07:07:58.711971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_test_eeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/3911565283.parquet')\n# sample_test_spec = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/853520.parquet')\n# print(sample_test_eeg.shape)\n# print(sample_test_spec.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['class name'] = df_train['expert_consensus'].copy()\ndf_train['expert_consensus'] = df_train['expert_consensus'].replace({'Seizure':1,'LPD':2,'GPD':3,'LRDA':4,'GRDA':5,'Other':6})","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:07:58.714471Z","iopub.execute_input":"2024-06-21T07:07:58.714820Z","iopub.status.idle":"2024-06-21T07:07:58.784828Z","shell.execute_reply.started":"2024-06-21T07:07:58.714793Z","shell.execute_reply":"2024-06-21T07:07:58.783642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique patients\nnum_patients = df_train['patient_id'].nunique()\nprint(f\"Number of unique patients in train dataset: {num_patients}\")\n\n# Number of unique EEG IDs\nnum_eeg_ids = df_train['eeg_id'].nunique()\nprint(f\"Number of unique EEG IDs in train dataset: {num_eeg_ids}\")\n\n# Number of unique spectogram IDs\nnum_spec_ids= df_train['spectrogram_id'].nunique()\nprint(f\"Number of unique spectogram IDs in train datasets: {num_spec_ids}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:07:58.785994Z","iopub.execute_input":"2024-06-21T07:07:58.786305Z","iopub.status.idle":"2024-06-21T07:07:58.798506Z","shell.execute_reply.started":"2024-06-21T07:07:58.786266Z","shell.execute_reply":"2024-06-21T07:07:58.797296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_rows = df_train.drop_duplicates(subset=['eeg_id', 'spectrogram_id'])\nunique_rows = unique_rows.reset_index(drop = True)\nunique_rows[['eeg_id','spectrogram_id','class name']].head()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:24:52.504566Z","iopub.execute_input":"2024-06-21T06:24:52.504990Z","iopub.status.idle":"2024-06-21T06:24:52.534263Z","shell.execute_reply.started":"2024-06-21T06:24:52.504950Z","shell.execute_reply":"2024-06-21T06:24:52.533181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Undersampling","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom sklearn.datasets import make_classification\nfrom imblearn.under_sampling import RandomUnderSampler\n\nrus = RandomUnderSampler(random_state=42)\nX = unique_rows.drop(columns = ['class name'])\ny = unique_rows[['class name']]\nX_res, y_res = rus.fit_resample(X,y)\n\nunique_rows = X_res.join(y_res)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:24:52.536013Z","iopub.execute_input":"2024-06-21T06:24:52.536428Z","iopub.status.idle":"2024-06-21T06:24:52.617569Z","shell.execute_reply.started":"2024-06-21T06:24:52.536388Z","shell.execute_reply":"2024-06-21T06:24:52.616442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting value counts\nvalue_counts = unique_rows['class name'].value_counts()\ntotal_counts = value_counts.sum()\n\n# Plotting the bar plot\nplt.figure(figsize=(10, 6))\nbar_plot = value_counts.plot(kind='bar', color='skyblue')\nplt.title('Expert Consensus Value Counts')\nplt.xlabel('Expert Consensus')\nplt.ylabel('Count')\nplt.xticks(rotation=0)\n\n# Adding the counts and percentages on top of each bar\nfor i in bar_plot.patches:\n    height = i.get_height()\n    percent = (height / total_counts) * 100\n    bar_plot.annotate(f'{height}\\n({percent:.1f}%)', \n                      (i.get_x() + i.get_width() / 2, height / 2), \n                      ha='center', va='center', fontsize=12, color='black')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:27:06.009573Z","iopub.execute_input":"2024-06-21T06:27:06.010526Z","iopub.status.idle":"2024-06-21T06:27:06.290851Z","shell.execute_reply.started":"2024-06-21T06:27:06.010477Z","shell.execute_reply":"2024-06-21T06:27:06.289866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def take_data(ids,slug):\n    # Cast all float columns to float16\n    df =pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_{slug}/{ids}.parquet')\n    if slug == 'eegs':\n        df = df.drop(columns = ['EKG'])\n    return df\n\nsample_eeg = {'Seizure': take_data(1628180742, 'eegs'), \n              'LPD': take_data(736446371,'eegs'), \n              'GRDA': take_data(936779698, 'eegs'), \n              'GPD': take_data(3172577040, 'eegs'), \n              'LRDA': take_data(4033061609, 'eegs')}\nsample_spectrogram = {'Seizure': take_data(353733, 'spectrograms'), \n                      'LPD': take_data(10397461, 'spectrograms'), \n                      'GRDA': take_data(15420126, 'spectrograms'), \n                      'GPD': take_data(20031028, 'spectrograms'), \n                      'LRDA': take_data(18324339, 'spectrograms')}\neeg_columns = ['Fp1','F3','C3','P3','F7','T3','T5','O1','Fz','Cz','Pz','Fp2','F4','C4','P4','F8','T4','T6','O2','EKG']","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:27:06.829935Z","iopub.execute_input":"2024-06-21T06:27:06.830383Z","iopub.status.idle":"2024-06-21T06:27:07.362013Z","shell.execute_reply.started":"2024-06-21T06:27:06.830349Z","shell.execute_reply":"2024-06-21T06:27:07.360883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_sample(data, col = None):\n    # Plotting each feature in subplots\n    if col is None:\n        col = list(data.columns)\n    else:\n        col = list(col)\n    fig, axes = plt.subplots(nrows=len(col), ncols=1, figsize=(20, 10))\n    fig.suptitle('EEG Data Visualization', fontsize=20, y=0.9)\n\n    # Adjusting spacing between subplots\n    plt.subplots_adjust(hspace=0.5)\n\n    # Plot each feature\n    for i, column in enumerate(col):\n        ax = axes[i]\n        ax.plot(data.index, data[column], marker='o', label=column, color='b', linestyle='-')\n        ax.set_xlabel('Sample', fontsize=14)\n        ax.set_ylabel('Amplitude', fontsize=14)\n        ax.set_title(column, fontsize=16, fontweight='bold')\n        ax.grid(True)\n        ax.legend(fontsize=12)\n        \n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:27:12.367328Z","iopub.execute_input":"2024-06-21T06:27:12.367745Z","iopub.status.idle":"2024-06-21T06:27:12.376522Z","shell.execute_reply.started":"2024-06-21T06:27:12.367711Z","shell.execute_reply":"2024-06-21T06:27:12.374971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr_matrix = spectrogram_data[0].iloc[:,:20].corr()\ndef heat_map(corr_matrix):\n# Create a heatmap using seaborn\n    plt.figure(figsize=(20, 16))\n    sns.heatmap(corr_matrix, annot=True, fmt=\".2f\", cmap='coolwarm', cbar=True)\n    plt.title('Correlation Matrix Heatmap for EEG Data')\n    plt.show()\nheat_map(corr_matrix)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mne\ndef preprocess(data):\n    # Handle missing values (if any)\n    # Here, we fill missing values with the mean of the respective columns\n    \n    # Clip values to reduce the effect of outliers (if needed)\n    # Here, we clip values outside of 2nd and 98th percentiles\n    lower_percentile = data.quantile(0.02)\n    upper_percentile = data.quantile(0.98)\n    data = data.clip(lower=lower_percentile, upper=upper_percentile, axis=1)\n    # Perform FFT on each row (assuming data is in a DataFrame format)\n    return data\ndef pad_or_truncate_eeg(eeg_data, target_length=1000):\n#     eeg_data = eeg_data.to_numpy()\n    current_length, features = eeg_data.shape\n#     print(current_length)\n    if current_length < target_length:\n        # Pad with zeros\n        padding = np.zeros((target_length - current_length, features))\n        eeg_data_padded = np.vstack((eeg_data, padding))\n    elif current_length > target_length:\n        # Truncate to target length\n        eeg_data_padded = eeg_data[:target_length, :]\n    else:\n        eeg_data_padded = eeg_data\n    return eeg_data_padded\n\ndef pad_or_truncate_spec(spectrogram_data, target_time_dimension=500):\n    current_time_dim, frequency_dim = spectrogram_data.shape\n    spectrogram_data = spectrogram_data.to_numpy()\n    if current_time_dim < target_time_dimension:\n        # Pad with zeros\n        padding = np.zeros((target_time_dimension - current_time_dim, frequency_dim))\n        spectrogram_data_padded = np.vstack((spectrogram_data, padding))\n    elif current_time_dim > target_time_dimension:\n        # Truncate to target time dimension\n        spectrogram_data_padded = spectrogram_data[:target_time_dimension, :]\n    else:\n        spectrogram_data_padded = spectrogram_data\n    return spectrogram_data_padded","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:27:16.066303Z","iopub.execute_input":"2024-06-21T06:27:16.066735Z","iopub.status.idle":"2024-06-21T06:27:17.400410Z","shell.execute_reply.started":"2024-06-21T06:27:16.066695Z","shell.execute_reply":"2024-06-21T06:27:17.399327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\n# Define the function to process each row\nscaler = StandardScaler()\ndef process_eeg(row):\n    eeg_data = take_data(row['eeg_id'], 'eegs').replace({np.inf:np.nan,-np.inf:np.nan})\n    eeg_data = eeg_data.dropna()\n    eeg_data = preprocess(eeg_data)\n    eeg_data = scaler.fit_transform(eeg_data)\n    eeg_data = pad_or_truncate_eeg(eeg_data)\n    return eeg_data\ndef process_spec(row):\n    spec_data = take_data(row['spectrogram_id'], 'spectrograms')\n    spec_data = spec_data.drop(columns = ['time']).dropna()\n    spec_data = preprocess(spec_data)\n    spec_data = pad_or_truncate_spec(spec_data)\n    return spec_data\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:27:38.280040Z","iopub.execute_input":"2024-06-21T06:27:38.280972Z","iopub.status.idle":"2024-06-21T06:27:38.288190Z","shell.execute_reply.started":"2024-06-21T06:27:38.280928Z","shell.execute_reply":"2024-06-21T06:27:38.286947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = unique_rows.iloc[2]\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:29:32.985086Z","iopub.execute_input":"2024-06-21T06:29:32.985525Z","iopub.status.idle":"2024-06-21T06:29:33.128505Z","shell.execute_reply.started":"2024-06-21T06:29:32.985493Z","shell.execute_reply":"2024-06-21T06:29:33.126864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use joblib to parallelize the processing of rows\ndata_eeg = np.array(joblib.Parallel(n_jobs=-1, backend ='loky',return_as ='list')(\n    joblib.delayed(process_eeg)(unique_rows.iloc[row]) for row in tqdm(range(len(unique_rows)))\n)).astype(np.float16)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = unique_rows.iloc[2221]\ndf_sample = take_data(data['eeg_id'],'eegs')\nscalled = scaler.fit_transform(df_sample)\nmean = scaler.mean_\nprint(mean)\nstd =scaler.var_\ncv = mean/std*100\nprint(cv)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:47.600721Z","iopub.execute_input":"2024-06-21T06:50:47.601211Z","iopub.status.idle":"2024-06-21T06:50:47.631479Z","shell.execute_reply.started":"2024-06-21T06:50:47.601171Z","shell.execute_reply":"2024-06-21T06:50:47.630114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:53.822258Z","iopub.execute_input":"2024-06-21T06:50:53.822632Z","iopub.status.idle":"2024-06-21T06:50:53.850795Z","shell.execute_reply.started":"2024-06-21T06:50:53.822604Z","shell.execute_reply":"2024-06-21T06:50:53.849761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = unique_rows.iloc[]\ndf_sample = take_data(data['eeg_id'],'eegs')\nscalled = scaler.fit_transform(df_sample)\nmean = scaler.mean_\nprint(mean)\nstd =scaler.var_\ncv = mean/std*100\nprint(cv)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:52:04.441705Z","iopub.execute_input":"2024-06-21T06:52:04.442144Z","iopub.status.idle":"2024-06-21T06:52:04.484725Z","shell.execute_reply.started":"2024-06-21T06:52:04.442112Z","shell.execute_reply":"2024-06-21T06:52:04.483547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_labels","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:55:07.058400Z","iopub.execute_input":"2024-06-21T06:55:07.058832Z","iopub.status.idle":"2024-06-21T06:55:07.102940Z","shell.execute_reply.started":"2024-06-21T06:55:07.058798Z","shell.execute_reply":"2024-06-21T06:55:07.101462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use joblib to parallelize the processing of rows\ndata_spec = np.array(joblib.Parallel(n_jobs=-1, backend ='loky',return_as ='list')(\n    joblib.delayed(process_spec)(unique_rows.iloc[row]) for row in tqdm(range(len(unique_rows)))\n)).","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py\n\n# Save the arrays to an HDF5 file\nwith h5py.File('eeg_data_10000_.h5', 'w') as f:\n    f.create_dataset('eeg_data', data=data_eeg, compression='gzip')\n\nwith h5py.File('spec_data.h5', 'w') as f:\n    f.create_dataset('spec_data', data=data_spec, compression='gzip')\n    \n# os.remove(\"/kaggle/working/eeg_data_10000_.npz\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py\nwith h5py.File('/kaggle/working/eeg_data_10000_.h5', 'r') as f:\n    eeg_data = f['eeg_data'][:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = unique_rows[['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']].to_numpy()\nohe_labels = np.argmax(labels,axis =1)\nfinal_labels = to_categorical(ohe_labels)\nfinal_labels.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-20T23:15:50.883836Z","iopub.execute_input":"2024-06-20T23:15:50.884349Z","iopub.status.idle":"2024-06-20T23:15:50.896563Z","shell.execute_reply.started":"2024-06-20T23:15:50.884296Z","shell.execute_reply":"2024-06-20T23:15:50.895221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, LSTM, Dense, concatenate\n# Define the input layers\neeg_input = Input(shape=(25000, 20), name='eeg_input')\nspectrogram_input = Input(shape=(900, 400), name='spectrogram_input')\n\n# Define the LSTM layers\neeg_lstm = LSTM(1024)(eeg_input)\nspectrogram_lstm = LSTM(512)(spectrogram_input)\n\n# Concatenate the outputs of the LSTM layers\nconcatenated = concatenate([eeg_lstm, spectrogram_lstm])\n\n# Add a fully connected layer and output layer\ndense = Dense(64, activation='relu')(concatenated)\noutput = Dense(6, activation='softmax')(dense)\n\n# Define the model\nmodel = Model(inputs=[eeg_input, spectrogram_input], outputs=output)\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    [eeg_data_padded, spectrogram_data_padded], \n    final_labels, \n    epochs=5, \n    batch_size= 11, \n    validation_split=0.15,\n    verbose = 1\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EfficientNet5B","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, Conv2D, Reshape, Permute\nfrom tensorflow.keras.applications import EfficientNetB5\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import Callback\nLOSS =  keras.losses.KLDivergence()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T23:22:24.916005Z","iopub.execute_input":"2024-06-20T23:22:24.916644Z","iopub.status.idle":"2024-06-20T23:22:24.930415Z","shell.execute_reply.started":"2024-06-20T23:22:24.916593Z","shell.execute_reply":"2024-06-20T23:22:24.928947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    # Split data into training and validation sets\n    X_train, X_val, y_train, y_val = train_test_split(data_eeg, final_labels, test_size=0.15, random_state=42)\n\n    # Define the custom input layer for EEG data\n    input_shape = (1000, 19,1)\n    inputs = Input(shape=input_shape)\n\n    # Reshape or preprocess the data to fit EfficientNetB5 input requirements\n    # Example: Resizing the input to (100, 100, 3) to meet the minimum requirement of 32x32\n    x = Conv2D(3, kernel_size=(3, 3), padding='same')(inputs)  # Convert to 3 channels\n    x = Reshape((500, 38, 3))(x)  # Reshape to (height, width, channels)\n    x = Permute((2, 1, 3))(x)  # Permute dimensions to match (width, height, channels)\n\n    # Load the pretrained EfficientNetB5 model\n    base_model = EfficientNetB5(weights='imagenet', include_top=False, input_shape=(38, 500, 3))\n    x = base_model(x)\n\n    # Add custom layers on top\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dense(6, activation='linear')(x)  # Adjust the number of classes as needed\n    x = tf.keras.layers.Lambda(lambda x: x + 0.01)(x)\n    predictions = tf.keras.activations.softmax(x)\n\n    # Define the model\n    model = Model(inputs=inputs, outputs=predictions)\n\n    # Compile the model\n    model.compile(optimizer= Adam(learning_rate = 0.00001), loss=LOSS, metrics=['accuracy'])\n    \n    # Train the motdel\n    history = model.fit(X_train, y_train, epochs=10, batch_size=2, validation_data=(X_val, y_val))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T00:27:07.855790Z","iopub.execute_input":"2024-06-21T00:27:07.856329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNN-LSTM","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv1D, MaxPooling1D, LSTM, Dense, Dropout, Flatten, TimeDistributed\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom tensorflow.keras.optimizers import Adam\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(data_eeg, final_labels, test_size=0.2, random_state=42)\n\n# Define the input layer\ninputs = Input(shape=(10002, 20))\n\n# Define the CNN layers\nx = Conv1D(filters=64, kernel_size=3, activation='relu')(inputs)\nx = MaxPooling1D(pool_size=2)(x)\nx = Dropout(0.2)(x)\nx = Conv1D(filters=128, kernel_size=3, activation='relu')(x)\nx = MaxPooling1D(pool_size=2)(x)\nx = Dropout(0.2)(x)\nx = Conv1D(filters=256, kernel_size=3, activation='relu')(x)\nx = MaxPooling1D(pool_size=2)(x)\nx = Dropout(0.2)(x)\n\n# Reshape the data to fit into LSTM\nx = TimeDistributed(Flatten())(x)\n\n# Define the LSTM layers\nx = LSTM(100, return_sequences=True)(x)\nx = Dropout(0.2)(x)\nx = LSTM(100)(x)\nx = Dropout(0.2)(x)\n\n# Define the fully connected layers\nx = Dense(128, activation='relu')(x)\nx = Dropout(0.2)(x)\npredictions = Dense(6, activation='softmax')(x)  # Adjust the number of classes as needed\n\n# Define the model\nmodel = Model(inputs=inputs, outputs=predictions)\n# Compile the model with gradient clipping\noptimizer = Adam(learning_rate=0.001, clipvalue=0.5)\n# Compile the model\nmodel.compile(optimizer=optimizer, loss= keras.losses.KLDivergence(), metrics=['accuracy'])\n# 'categorical_crossentropy keras.losses.KLDivergence()'\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(X_train, y_train, epochs=5, batch_size=2, validation_data=(X_val, y_val))\n\n# Save the model\n# model.save('cnn_lstm_eeg_model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EfficientNetB5 on pytorch","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\ndata = [[0, 0], [0, 0], [1, 1], [1, 1]]\nscaler = StandardScaler()\nprint(scaler.fit(data))\nStandardScaler()\nprint(scaler.mean_)\nprint(scaler.transform(data))\nprint(scaler.transform([[2, 2]]))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-20T23:34:20.404636Z","iopub.execute_input":"2024-06-20T23:34:20.405172Z","iopub.status.idle":"2024-06-20T23:34:20.423161Z","shell.execute_reply.started":"2024-06-20T23:34:20.405134Z","shell.execute_reply":"2024-06-20T23:34:20.421298Z"},"trusted":true},"execution_count":null,"outputs":[]}]}