{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8050321,"sourceType":"datasetVersion","datasetId":4747381}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>Introduction</h1>","metadata":{}},{"cell_type":"markdown","source":"<h3>In this notebook I will attempt to fit spectrogram data in order to predict harmful brain activity. I am using a convolutional neural network from the TensorFlow framework.</h3>","metadata":{}},{"cell_type":"markdown","source":"<h1>Prepare Training Data</h1>","metadata":{}},{"cell_type":"markdown","source":"<h2>Define Imports and Set Global Variables</h2>","metadata":{}},{"cell_type":"code","source":"import os # Using for directory operations\nimport tqdm # progress bars\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plot it\nimport librosa\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier # For initial testing\n\n# Competition directory for that houses all data.\nCOMPDIR = '/kaggle/input/hms-harmful-brain-activity-classification'\n\n# Dataset directory for additional spectrogram data. Courtesy of CHRIS DEOTTE.\nDATPDIR = '/kaggle/input/brain-spectrograms'\n\n# Reading the CSV file 'train.csv' located in the directory specified by COMPDIR\ntrain = pd.read_csv(os.path.join(COMPDIR, 'train.csv'))\n\n# Reading the CSV file 'test.csv' located in the directory specified by COMPDIR\ntest = pd.read_csv(os.path.join(COMPDIR, 'test.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:38:51.381137Z","iopub.execute_input":"2024-04-18T00:38:51.381613Z","iopub.status.idle":"2024-04-18T00:38:51.585024Z","shell.execute_reply.started":"2024-04-18T00:38:51.381582Z","shell.execute_reply":"2024-04-18T00:38:51.583767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Load Spectrogram Data</h2>","metadata":{}},{"cell_type":"code","source":"# Import spectrogram info\nspect_data = np.load(os.path.join(DATPDIR, 'specs.npy'), allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-04-17T04:22:41.975577Z","iopub.execute_input":"2024-04-17T04:22:41.976011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h5>Credit: Dataset compiled by Chris Deotte.</h5>","metadata":{}},{"cell_type":"markdown","source":"<h2>Spectrogram Data Visualized</h2>","metadata":{}},{"cell_type":"code","source":"print(spect_data)\nprint(type(spect_data))\nprint(len(spect_data))","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:27.705258Z","iopub.execute_input":"2024-04-17T03:41:27.705677Z","iopub.status.idle":"2024-04-17T03:41:27.711198Z","shell.execute_reply.started":"2024-04-17T03:41:27.705646Z","shell.execute_reply":"2024-04-17T03:41:27.710003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example data\nprint(spect_data[319287046])\nprint(spect_data[319287046].shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:27.713768Z","iopub.execute_input":"2024-04-17T03:41:27.714118Z","iopub.status.idle":"2024-04-17T03:41:27.736874Z","shell.execute_reply.started":"2024-04-17T03:41:27.714073Z","shell.execute_reply":"2024-04-17T03:41:27.735850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a spectrogram data sample\ndata = spect_data[440834944]\n\n# Example of plotting a sample spectrogram\nplt.figure(figsize=(10, 4))\nlibrosa.display.specshow(data, x_axis='time', y_axis='mel', fmax=8000)\nplt.colorbar(format='%+2.0f dB')\nplt.title('Sample Spectrogram')\nplt.tight_layout()\nplt.show()\n\n# Plot the array\n# plt.imshow(data, cmap='viridis')  # You can choose different colormaps\n# plt.colorbar()  # Add a colorbar to the plot\n# plt.title('Spectrogram')\n# plt.xlabel('Time (s)')\n# plt.ylabel('Frequency (Hz)')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:27.738468Z","iopub.execute_input":"2024-04-17T03:41:27.739385Z","iopub.status.idle":"2024-04-17T03:41:28.300786Z","shell.execute_reply.started":"2024-04-17T03:41:27.739353Z","shell.execute_reply":"2024-04-17T03:41:28.299151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the number of training data points\nnum_train_data_points = 106800  # 106800, all of the data\n\n# Creating empty NumPy arrays to store training data\nspec_ids = np.empty((0,), dtype=np.int64)\nexpert_cons = np.empty((0,))\n\nlast_spec_id = -1\n\n# Iterating over each training data point\nfor i in tqdm.tqdm(range(num_train_data_points)):\n    # Loading spectrogram data for specified ids\n    spec_id = train.loc[i, 'spectrogram_id']\n    expert_con = train.loc[i, 'expert_consensus']\n    \n    if (spec_id != last_spec_id):\n        # Adding the spectrogram id and expert consensus to the NumPy arrays if unique\n        spec_ids = np.append(spec_ids, spec_id)\n        expert_cons = np.append(expert_cons, expert_con)\n    \n    last_spec_id = spec_id","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:28.302444Z","iopub.execute_input":"2024-04-17T03:41:28.302989Z","iopub.status.idle":"2024-04-17T03:41:33.463249Z","shell.execute_reply.started":"2024-04-17T03:41:28.302853Z","shell.execute_reply":"2024-04-17T03:41:33.461809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the stored data (converted to NumPy arrays)\nprint(\"Spec IDs NumPy array:\")\nprint(spec_ids)\nprint(spec_ids.shape)\nprint(\"\\nExpert Consensus NumPy array:\")\nprint(expert_cons)\nprint(expert_cons.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:33.464480Z","iopub.execute_input":"2024-04-17T03:41:33.464806Z","iopub.status.idle":"2024-04-17T03:41:33.472159Z","shell.execute_reply.started":"2024-04-17T03:41:33.464778Z","shell.execute_reply":"2024-04-17T03:41:33.471127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Preprocess Data</h2>","metadata":{}},{"cell_type":"markdown","source":"<h4>Since X_train is a dataset of floats and y_train is a dataset of strings, we need to encode the labels into numeric form and split your data into training and validation sets.</h4>","metadata":{}},{"cell_type":"code","source":"# X_train\n# y_train","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:33.610509Z","iopub.execute_input":"2024-04-17T03:41:33.610795Z","iopub.status.idle":"2024-04-17T03:41:33.653116Z","shell.execute_reply.started":"2024-04-17T03:41:33.610772Z","shell.execute_reply":"2024-04-17T03:41:33.651360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode labels into numerical form\nlabel_encoder = LabelEncoder()\ny_train_encoded = label_encoder.fit_transform(y_train)\n\n# Split data into training and validation sets\nX_train, X_val, y_train_encoded, y_val_encoded = train_test_split(X_train, y_train_encoded, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T03:41:33.654464Z","iopub.status.idle":"2024-04-17T03:41:33.655253Z","shell.execute_reply.started":"2024-04-17T03:41:33.654945Z","shell.execute_reply":"2024-04-17T03:41:33.654970Z"},"trusted":true},"execution_count":null,"outputs":[]}]}