{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re\nfrom string import punctuation\nfrom nltk.tokenize import word_tokenize\nfrom keras.utils import to_categorical\nimport string\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split\nimport spacy\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report\nfrom sklearn.metrics import log_loss\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.over_sampling import SMOTE, SMOTENC\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.combine import SMOTEENN\nfrom collections import Counter\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport joblib\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n#         print(os.path.join(dirname, filename))\n\npd.set_option('display.max_rows', 200)\npd.set_option('display.max_columns', 150)\npd.set_option('display.width', 1000)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-23T11:28:32.161609Z","iopub.execute_input":"2024-06-23T11:28:32.162567Z","iopub.status.idle":"2024-06-23T11:28:52.746833Z","shell.execute_reply.started":"2024-06-23T11:28:32.162497Z","shell.execute_reply":"2024-06-23T11:28:52.745884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Base_Path = '/kaggle/input/hms-harmful-brain-activity-classification'\ndf_train = pd.read_csv(f'{Base_Path}/train.csv')\ndf_test = pd.read_csv(f'{Base_Path}/test.csv')\ndf_train.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T11:13:22.092675Z","iopub.execute_input":"2024-06-23T11:13:22.093396Z","iopub.status.idle":"2024-06-23T11:13:22.383948Z","shell.execute_reply.started":"2024-06-23T11:13:22.093365Z","shell.execute_reply":"2024-06-23T11:13:22.382691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"def take_data(ids,slug):\n    # Cast all float columns to float16\n    df =pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_{slug}/{ids}.parquet')\n    if slug == 'eegs':\n        df = df.drop(columns = ['EKG'])\n    return df\nimport mne\ndef preprocess(data):\n    # Handle missing values (if any)\n    # Here, we fill missing values with the mean of the respective columns\n    \n    # Clip values to reduce the effect of outliers (if needed)\n    # Here, we clip values outside of 2nd and 98th percentiles\n    lower_percentile = data.quantile(0.02)\n    upper_percentile = data.quantile(0.98)\n    data = data.clip(lower=lower_percentile, upper=upper_percentile, axis=1)\n    # Perform FFT on each row (assuming data is in a DataFrame format)\n    return data\ndef pad_or_truncate_eeg(eeg_data, target_length=10000):\n#     eeg_data = eeg_data.to_numpy()\n    current_length, features = eeg_data.shape\n#     print(current_length)\n    if current_length < target_length:\n        # Pad with zeros\n        padding = np.zeros((target_length - current_length, features))\n        eeg_data_padded = np.vstack((eeg_data, padding))\n    elif current_length > target_length:\n        # Truncate to target length\n        eeg_data_padded = eeg_data[:target_length, :]\n    else:\n        eeg_data_padded = eeg_data\n    return eeg_data_padded\n\ndef pad_or_truncate_spec(spectrogram_data, target_time_dimension=500):\n    current_time_dim, frequency_dim = spectrogram_data.shape\n    spectrogram_data = spectrogram_data.to_numpy()\n    if current_time_dim < target_time_dimension:\n        # Pad with zeros\n        padding = np.zeros((target_time_dimension - current_time_dim, frequency_dim))\n        spectrogram_data_padded = np.vstack((spectrogram_data, padding))\n    elif current_time_dim > target_time_dimension:\n        # Truncate to target time dimension\n        spectrogram_data_padded = spectrogram_data[:target_time_dimension, :]\n    else:\n        spectrogram_data_padded = spectrogram_data\n    return spectrogram_data_padded\nimport numpy as np\nimport pywt\n\ndef perform_cwt(signal, wavelet_width, fs, lower_freq, upper_freq, n_scales, border_crop, stride):\n    # Calculate the central frequencies of the wavelet\n    wavelet = 'cmor' + str(wavelet_width) + '-1.0'\n    central_freq = pywt.central_frequency(wavelet)\n\n    # Calculate scales based on the desired frequency range\n    scales = central_freq * fs / np.linspace(lower_freq, upper_freq, n_scales)\n    \n    # Perform the Continuous Wavelet Transform\n    coefs, freqs = pywt.cwt(signal, scales, wavelet, sampling_period=1.0/fs)\n    \n    # Crop borders if needed\n    if border_crop > 0:\n        coefs = coefs[:, border_crop:-border_crop]\n\n    # Apply stride\n    coefs = coefs[:, ::stride]\n    \n    return coefs, freqs\n\n# Define parameters\nwavelet_width = 7\nfs = 200\nlower_freq = 0.5\nupper_freq = 20\nn_scales = 40\nborder_crop = 1\nstride = 16\n\n# Perform CWT for each channel\ndef cwt_all(raw_eeg_signal):\n    cwt_results = []\n    for i in range(raw_eeg_signal.shape[1]):\n        signal = raw_eeg_signal[:, i]\n        coefs, freqs = perform_cwt(signal, wavelet_width, fs, lower_freq, upper_freq, n_scales, border_crop, stride)\n        cwt_results.append(coefs)\n    return np.abs(np.array(cwt_results))\n# cwt_results is a list of CWT results for each EEG channel\n# Create a figure and axis\n\ndef mat_show(img):\n    fig, ax = plt.subplots(figsize=(10, 6))\n\n    # Use matshow to visualize the magnitude of CWT coefficients of the specified channel\n    cax = ax.matshow(img, aspect='auto', cmap='viridis')\n\n    # Add a color bar to show the intensity scale\n    fig.colorbar(cax)\n\n    # Set labels\n    ax.set_title(f'CWT Magnitude for Channel')\n    ax.set_xlabel('Time')\n    ax.set_ylabel('Scale')\n\n    # Show the plot\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-23T10:08:39.448374Z","iopub.execute_input":"2024-06-23T10:08:39.449050Z","iopub.status.idle":"2024-06-23T10:08:40.175876Z","shell.execute_reply.started":"2024-06-23T10:08:39.449015Z","shell.execute_reply":"2024-06-23T10:08:40.175039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nimport numpy as np\nimport cv2\n# Define the function to process each row\nscaler = StandardScaler()\ndef process_eeg(row):\n    sample_signal = take_data(row['eeg_id'], 'eegs').replace({np.inf:np.nan,-np.inf:np.nan})\n    sample_signal = sample_signal.dropna()\n    mapped_features = np.array([sample_signal['Fp1'] - sample_signal['F7'],\n                                sample_signal['F7'] - sample_signal['T3'],\n                                sample_signal['T3'] - sample_signal['T5'],\n                                sample_signal['T5'] - sample_signal['O1'],\n                                sample_signal['Fp1'] - sample_signal['F3'],\n                                sample_signal['F3'] - sample_signal['C3'],\n                                sample_signal['C3'] - sample_signal['P3'],\n                                sample_signal['P3'] - sample_signal['O1'],\n                                sample_signal['Fp2'] - sample_signal['F8'],\n                                sample_signal['F8'] - sample_signal['T4'],\n                                sample_signal['T4'] - sample_signal['T6'],\n                                sample_signal['T6'] - sample_signal['O2'],\n                                sample_signal['Fp2'] - sample_signal['F4'],\n                                sample_signal['F4'] - sample_signal['C4'],\n                                sample_signal['C4'] - sample_signal['P4'],\n                                sample_signal['P4'] - sample_signal['O2'],\n                                sample_signal['Fz'] - sample_signal['Cz'],\n                                sample_signal['Cz'] - sample_signal['Pz']])\n    mapped_features = np.transpose(mapped_features)\n    scalogram_image = cwt_all(mapped_features)\n    # Step 1: Stack the first two dimensions (18 and 40) into one dimension\n    stacked_image = np.reshape(scalogram_image, (18 * 40, scalogram_image.shape[2]))  # Resulting shape: (720, 625)\n\n    # Step 2: Resize the stacked image to (512, 512)\n    scalogram_image = cv2.resize(stacked_image, (512, 512), interpolation=cv2.INTER_LINEAR)\n    # stft wavelet\n    return scalogram_image","metadata":{"execution":{"iopub.status.busy":"2024-06-23T10:08:40.177815Z","iopub.execute_input":"2024-06-23T10:08:40.178906Z","iopub.status.idle":"2024-06-23T10:08:40.361398Z","shell.execute_reply.started":"2024-06-23T10:08:40.178868Z","shell.execute_reply":"2024-06-23T10:08:40.360405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unique Rows","metadata":{}},{"cell_type":"code","source":"Base_Path = '/kaggle/input/hms-harmful-brain-activity-classification'\ndf_train = pd.read_csv(f'{Base_Path}/train.csv')\ndf_train['class name'] = df_train['expert_consensus'].copy()\ndf_train['expert_consensus'] = df_train['expert_consensus'].replace({'Seizure':1,'LPD':2,'GPD':3,'LRDA':4,'GRDA':5,'Other':6})\nunique_rows = df_train.drop_duplicates(subset=['eeg_id', 'spectrogram_id'])\nunique_rows[['eeg_id','spectrogram_id','class name']].head()\nunique_rows = unique_rows[unique_rows[['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote']].T.sum() > 5]\nunique_rows = unique_rows.reset_index(drop = True)\nprint(len(unique_rows))\nlabels = unique_rows[['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']].to_numpy()\nohe_labels = np.argmax(labels,axis =1)\nfinal_labels = to_categorical(ohe_labels)\nprint(final_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T11:29:40.324663Z","iopub.execute_input":"2024-06-23T11:29:40.325445Z","iopub.status.idle":"2024-06-23T11:29:40.688468Z","shell.execute_reply.started":"2024-06-23T11:29:40.325410Z","shell.execute_reply":"2024-06-23T11:29:40.687390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_at = 13\nmat_show(process_eeg(unique_rows.iloc[sample_at]))\nprint(unique_rows['class name'].iloc[sample_at])\ndata = process_eeg(unique_rows.iloc[sample_at])\nprint(np.max(data), np.min(data))","metadata":{"execution":{"iopub.status.busy":"2024-06-23T09:56:10.000539Z","iopub.execute_input":"2024-06-23T09:56:10.000940Z","iopub.status.idle":"2024-06-23T09:56:56.650382Z","shell.execute_reply.started":"2024-06-23T09:56:10.000905Z","shell.execute_reply":"2024-06-23T09:56:56.649395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Extraction","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\n# Use joblib to parallelize the processing of rows\ndata_eeg = np.array(joblib.Parallel(n_jobs=-1, backend ='loky',return_as ='list')(\n    joblib.delayed(process_eeg)(unique_rows.iloc[row]) for row in tqdm(range(len(unique_rows))) # len(unique_rows)\n)).astype(np.float16)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T10:08:54.178710Z","iopub.execute_input":"2024-06-23T10:08:54.179099Z","iopub.status.idle":"2024-06-23T11:02:02.664314Z","shell.execute_reply.started":"2024-06-23T10:08:54.179070Z","shell.execute_reply":"2024-06-23T11:02:02.663233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save to a .npy file\n# np.save('5-3195.npy', data_eeg)\n# Define file path\nfile_path = '/kaggle/working/5-3195.npy'\n\n# Save the array\nnp.save(file_path, data_eeg)\n# np.save('kaggle/working/final_labels-3195.npy',final_labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T11:05:56.468574Z","iopub.execute_input":"2024-06-23T11:05:56.468958Z","iopub.status.idle":"2024-06-23T11:05:57.875428Z","shell.execute_reply.started":"2024-06-23T11:05:56.468931Z","shell.execute_reply":"2024-06-23T11:05:57.874555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load data","metadata":{}},{"cell_type":"code","source":"# Load a .npz file\ndata_eeg = np.load('/kaggle/working/5-3195.npy')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Indices of the rows to drop\nindices_to_drop = [1439, 1440, 2764]\n\n# Remove the specified rows\ndata_eeg = np.delete(data_eeg, indices_to_drop, axis=0)\n\ninf_mask = np.isinf(data_eeg)\n# Find the indices of the `inf` values\ninf_indices = np.argwhere(inf_mask)\nprint(inf_indices)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.max(data_eeg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_labels = np.delete(final_labels, [1439,1440,2764], axis = 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n\n# # Expand dimensions to (50, 512, 512, 1)\n# input_data_expanded = np.expand_dims(data_eeg, axis=-1)\n\n# # Repeat along the channel axis to create three channels\n# input_data_repeated = np.repeat(input_data_expanded, 3, axis=-1)\n\n# # Now `input_data_repeated` has shape (50, 512, 512, 3)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T11:13:29.756134Z","iopub.execute_input":"2024-06-23T11:13:29.757064Z","iopub.status.idle":"2024-06-23T11:13:35.464849Z","shell.execute_reply.started":"2024-06-23T11:13:29.757020Z","shell.execute_reply":"2024-06-23T11:13:35.463344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Efficient Net B5","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB5\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Input, Lambda\nfrom tensorflow.keras.models import Model\nfrom tensorflow import keras\nLOSS =  keras.losses.KLDivergence()\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(data_eeg, final_labels, test_size=0.15, random_state=42)\n\n# Define the input layer with shape (512, 512)\ninput_layer = Input(shape=(512, 512))\n\n# Define a Lambda layer to expand and repeat the dimensions\nexpanded_layer = Lambda(lambda x: tf.repeat(tf.expand_dims(x, axis=-1), 3, axis=-1))(input_layer)\n# Load pre-trained EfficientNetB5 model without the top classification layer\nbase_model = EfficientNetB5(weights='imagenet', include_top=False, input_shape=(512, 512, 3),input_tensor=expanded_layer)\n\n# Add your own classification layer\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\npredictions = Dense(6, activation='softmax')(x)\n\n# Combine base model and custom layers into a new model\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss=LOSS, metrics=['accuracy'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optional: Freeze initial layers of the base model\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# Train the motdel\nhistory = model.fit(X_train, y_train, epochs=30, batch_size=2, validation_data=(X_val, y_val))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}