{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8150933,"sourceType":"datasetVersion","datasetId":4820683},{"sourceId":8189363,"sourceType":"datasetVersion","datasetId":4849518}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>Introduction</h1>","metadata":{}},{"cell_type":"markdown","source":"<h3>In this notebook I will attempt to fit electroencephalogram data in order to predict harmful brain activity. I am using a convolutional neural network from the TensorFlow framework.</h3>","metadata":{}},{"cell_type":"markdown","source":"<h1>Prepare Training Data</h1>","metadata":{}},{"cell_type":"markdown","source":"<h2>Define Imports and Set Global Variables</h2>","metadata":{}},{"cell_type":"code","source":"import os # Using for directory operations\nimport tqdm # progress bars\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plot it\nimport librosa\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier # For initial testing\n\n# Competition directory for that houses all data.\nCOMPDIR = '/kaggle/input/hms-harmful-brain-activity-classification'\n\n# Need to retest 25000 samples\n# EEG_RETEST = '/kaggle/input/eeg-retest'\n\n# Test it all\nEEG_FULLTEST = '/kaggle/input/eeg-fulltest'\n\n# Reading the CSV file 'train.csv' located in the directory specified by COMPDIR\ntrain = pd.read_csv(os.path.join(COMPDIR, 'train.csv'))\n\n# Reading the CSV file 'test.csv' located in the directory specified by COMPDIR\ntest = pd.read_csv(os.path.join(COMPDIR, 'test.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:39:22.779515Z","iopub.execute_input":"2024-04-22T20:39:22.780042Z","iopub.status.idle":"2024-04-22T20:39:22.993002Z","shell.execute_reply.started":"2024-04-22T20:39:22.780004Z","shell.execute_reply":"2024-04-22T20:39:22.991577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import retest info\n# eeg_data = np.load(os.path.join(EEG_RETEST, 'eeg_retest.npy'), allow_pickle=True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-22T20:39:22.995002Z","iopub.execute_input":"2024-04-22T20:39:22.995372Z","iopub.status.idle":"2024-04-22T20:39:23.001039Z","shell.execute_reply.started":"2024-04-22T20:39:22.995340Z","shell.execute_reply":"2024-04-22T20:39:22.999702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import retest info\neeg_data = np.load(os.path.join(EEG_FULLTEST, 'eeg_fulltest.npy'), allow_pickle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:39:23.002397Z","iopub.execute_input":"2024-04-22T20:39:23.002818Z","iopub.status.idle":"2024-04-22T20:40:01.797403Z","shell.execute_reply.started":"2024-04-22T20:39:23.002759Z","shell.execute_reply":"2024-04-22T20:40:01.796090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the number of training data points\nnum_train_data_points = 106800  # 106800, all of the data\n\n# Creating empty NumPy arrays to store training data\nexpert_cons = np.empty((0,))\n\nlast_spec_id = -1\n\n# Iterating over each training data point\nfor i in tqdm.tqdm(range(num_train_data_points)):\n    # Loading spectrogram data for specified ids\n    expert_con = train.loc[i, 'expert_consensus']\n    expert_cons = np.append(expert_cons, expert_con)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:40:01.798729Z","iopub.execute_input":"2024-04-22T20:40:01.799210Z","iopub.status.idle":"2024-04-22T20:43:48.241996Z","shell.execute_reply.started":"2024-04-22T20:40:01.799168Z","shell.execute_reply":"2024-04-22T20:43:48.240747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(expert_cons.shape)\nprint(expert_cons)\n\nnan_indices = np.argwhere(np.isnan(eeg_data))\neeg_data_clean = eeg_data[~np.isnan(eeg_data).any(axis=1)]\n\n# Assuming another_array is the array from which you want to remove rows corresponding to NaN indices\nexpert_cons_clean = expert_cons[~np.isnan(eeg_data).any(axis=1)]\n\nprint(\"Indices of NaNs:\", nan_indices)\nprint(\"Data after removing NaNs:\")\nprint(eeg_data_clean)\nprint(eeg_data_clean.shape)\nprint(\"Data after removing corresponding rows:\")\nprint(expert_cons_clean)\nprint(expert_cons_clean.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:48.246137Z","iopub.execute_input":"2024-04-22T20:43:48.246573Z","iopub.status.idle":"2024-04-22T20:43:55.168713Z","shell.execute_reply.started":"2024-04-22T20:43:48.246537Z","shell.execute_reply":"2024-04-22T20:43:55.167613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = eeg_data_clean\ny_train = expert_cons_clean","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:55.170101Z","iopub.execute_input":"2024-04-22T20:43:55.170516Z","iopub.status.idle":"2024-04-22T20:43:55.176180Z","shell.execute_reply.started":"2024-04-22T20:43:55.170482Z","shell.execute_reply":"2024-04-22T20:43:55.174821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Preprocess Data</h2>","metadata":{}},{"cell_type":"markdown","source":"<h4>Since X_train is a dataset of floats and y_train is a dataset of strings, we need to encode the labels into numeric form and split your data into training and validation sets.</h4>","metadata":{}},{"cell_type":"code","source":"# Encode labels into numerical form\nlabel_encoder = LabelEncoder()\ny_train_encoded = label_encoder.fit_transform(y_train)\n\n# Split data into training and validation sets\nX_train, X_val, y_train_encoded, y_val_encoded = train_test_split(X_train, y_train_encoded, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:55.178110Z","iopub.execute_input":"2024-04-22T20:43:55.178467Z","iopub.status.idle":"2024-04-22T20:43:56.961320Z","shell.execute_reply.started":"2024-04-22T20:43:55.178434Z","shell.execute_reply":"2024-04-22T20:43:56.960145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train_encoded)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:56.962599Z","iopub.execute_input":"2024-04-22T20:43:56.962969Z","iopub.status.idle":"2024-04-22T20:43:56.969028Z","shell.execute_reply.started":"2024-04-22T20:43:56.962938Z","shell.execute_reply":"2024-04-22T20:43:56.967662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Define and Compile CNN Model</h2>","metadata":{}},{"cell_type":"code","source":"input_shape = X_train.shape[1:]\n\nmodel = models.Sequential([\n    layers.Reshape(input_shape=input_shape + (1,), target_shape=input_shape + (1,)),  # Reshape for 1D Conv input\n    layers.Conv1D(32, kernel_size=3, activation='relu'),\n    layers.MaxPooling1D(pool_size=2),\n    layers.Conv1D(64, kernel_size=3, activation='relu'),\n    layers.MaxPooling1D(pool_size=2),\n    layers.Flatten(),\n    layers.Dense(128, activation='relu'),\n    layers.Dense(len(np.unique(y_train_encoded)), activation='softmax')  # Output layer\n])\n\nmodel.summary()\n\nmodel.compile(optimizer='adam',\n          loss='sparse_categorical_crossentropy',\n          metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:56.970657Z","iopub.execute_input":"2024-04-22T20:43:56.971295Z","iopub.status.idle":"2024-04-22T20:43:57.308131Z","shell.execute_reply.started":"2024-04-22T20:43:56.971250Z","shell.execute_reply":"2024-04-22T20:43:57.306865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Train Model</h2>","metadata":{}},{"cell_type":"code","source":"history = model.fit(X_train, y_train_encoded, epochs=10, batch_size=32, validation_data=(X_val, y_val_encoded))","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:43:57.309728Z","iopub.execute_input":"2024-04-22T20:43:57.310152Z","iopub.status.idle":"2024-04-22T20:44:15.432256Z","shell.execute_reply.started":"2024-04-22T20:43:57.310119Z","shell.execute_reply":"2024-04-22T20:44:15.429849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation loss values\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(['Train', 'Validation'], loc='upper right')\nplt.show()\n\n# Plot training & validation accuracy values\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend(['Train', 'Validation'], loc='lower right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:44:15.433932Z","iopub.status.idle":"2024-04-22T20:44:15.434372Z","shell.execute_reply.started":"2024-04-22T20:44:15.434174Z","shell.execute_reply":"2024-04-22T20:44:15.434191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1>Prepare Test Data</h1>","metadata":{}},{"cell_type":"markdown","source":"<h2>Prepare Features (X_test)</h2>","metadata":{}},{"cell_type":"code","source":"# Creating an empty DataFrame to store testing data\nX_test = pd.DataFrame()\n\n# Iterating over each test data point\nfor i in tqdm.tqdm(range(len(test))):\n    # Loading EEG data for a specified eeg_id\n    eeg_id_ = test.loc[i, 'eeg_id']\n    tmp = pd.read_parquet(os.path.join(COMPDIR, 'test_eegs', f'{eeg_id_}.parquet'))\n    \n    # Extracting EEG data from the Cz electrode\n    cz_electrode_data = tmp['Cz']\n    \n    # Adding the extracted data as a row to the testing DataFrame\n    X_test = pd.concat([X_test, cz_electrode_data.reset_index(drop=True).to_frame().transpose()], axis=0)\n    \ndisplay(cz_electrode_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:44:15.436093Z","iopub.status.idle":"2024-04-22T20:44:15.436551Z","shell.execute_reply.started":"2024-04-22T20:44:15.436323Z","shell.execute_reply":"2024-04-22T20:44:15.436339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Predict and Submit</h2>","metadata":{}},{"cell_type":"code","source":"# Assuming label_encoder is your LabelEncoder instance used during training\nclass_labels = label_encoder.inverse_transform(np.arange(len(label_encoder.classes_)))\n\n# Calculate predictions using the trained RandomForestClassifier model\npredictions = model.predict(X_test)\n\n# Read the sample submission file\nsubmission = pd.read_csv(f'{COMPDIR}/sample_submission.csv')\n\n# Iterate over each test data point\nfor i in tqdm.tqdm(range(len(test))):\n    # Set the 'eeg_id' in the submission DataFrame\n    submission.loc[i, 'eeg_id'] = test.loc[i, 'eeg_id']\n    \n    # Set the probability for each class in the submission DataFrame\n    for j, cls_name in enumerate(class_labels):\n        # Note: This actually performs better when introducing pseudo-random data.\n        #submission.loc[i, f'{cls_name.lower()}_vote'] = random_variables[j]        \n        submission.loc[i, f'{cls_name.lower()}_vote'] = predictions[i, j]","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:44:15.439223Z","iopub.status.idle":"2024-04-22T20:44:15.439907Z","shell.execute_reply.started":"2024-04-22T20:44:15.439509Z","shell.execute_reply":"2024-04-22T20:44:15.439532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the submission DataFrame\ndisplay(submission)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:44:15.441109Z","iopub.status.idle":"2024-04-22T20:44:15.441525Z","shell.execute_reply.started":"2024-04-22T20:44:15.441324Z","shell.execute_reply":"2024-04-22T20:44:15.441340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the submission DataFrame to a CSV file without including the index\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T20:44:15.443623Z","iopub.status.idle":"2024-04-22T20:44:15.444128Z","shell.execute_reply.started":"2024-04-22T20:44:15.443917Z","shell.execute_reply":"2024-04-22T20:44:15.443937Z"},"trusted":true},"execution_count":null,"outputs":[]}]}