{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"47386baa-52ec-42c9-b6dc-b1e6afe90aef","_cell_guid":"5a002121-d1b3-4efb-a2aa-e11d36f5decd","trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"execution":{"iopub.status.busy":"2025-02-26T23:46:04.034194Z","iopub.execute_input":"2025-02-26T23:46:04.034635Z","iopub.status.idle":"2025-02-26T23:46:39.188353Z","shell.execute_reply.started":"2025-02-26T23:46:04.034600Z","shell.execute_reply":"2025-02-26T23:46:39.187276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# What is Electroencephalography (EEG)?\nElectroencephalography (EEG) is a medical technique used to measure and record the electrical activity of the brain. It works by placing small electrodes on the scalp to detect neural signals generated by brain cells.\n\n# Why is EEG Used?\nEEG is widely used in various medical and research applications, including:\n\n✅ Diagnosing Epileptic Seizures – Helps detect abnormal brain activity patterns associated with seizures.\n\n✅ Sleep Disorder Analysis – Monitors brain activity during sleep to identify issues like insomnia or sleep apnea.\n\n✅ Brain Function Research – Used in neuroscience studies to understand how the brain functions in both healthy and diseased states.\n\n✅ Monitoring Stroke or Head Injury Patients – Assists in evaluating brain function after trauma.\n\n# How Does EEG Work?\nEEG measures voltage fluctuations caused by brain activity and represents them as waveforms. These waveforms help doctors and researchers analyze different brain states, such as wakefulness, sleep, and neurological disorders.\n\n\n# Types of Brain Waves Recorded by EEG\n\nEEG detects different types of brain waves, classified based on their frequency:\n\nDelta Waves (0.5 - 4 Hz): Associated with deep sleep.\n\nTheta Waves (4 - 8 Hz): Related to relaxation and light sleep.\n\nAlpha Waves (8 - 12 Hz): Occur during calm, resting states.\n\nBeta Waves (12 - 30 Hz): Linked to active thinking and focus.\n\nGamma Waves (30+ Hz): Associated with high-level cognitive functioning.\n\n# EEG Recording System: The 10-20 System\n\nThe 10-20 system is an internationally recognized method for placing electrodes on the scalp. It ensures consistent placement of sensors to provide accurate and reproducible recordings. Electrodes are named based on the brain region they monitor:\n\nF (Frontal): Decision-making and problem-solving.\n\nT (Temporal): Memory and auditory processing.\n\nC (Central): Motor function.\n\nP (Parietal): Sensory processing.\n\nO (Occipital): Vision and image processing.\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nt = np.linspace(0, 1, 500)  \n\ndelta_wave = np.sin(2 * np.pi * 2 * t)  \ntheta_wave = np.sin(2 * np.pi * 6 * t) \nalpha_wave = np.sin(2 * np.pi * 10 * t) \nbeta_wave = np.sin(2 * np.pi * 20 * t)  \ngamma_wave = np.sin(2 * np.pi * 40 * t)\n\nplt.figure(figsize=(10,6))\nplt.plot(t, delta_wave, label=\"Delta (0.5-4 Hz)\")\nplt.plot(t, theta_wave, label=\"Theta (4-8 Hz)\")\nplt.plot(t, alpha_wave, label=\"Alpha (8-12 Hz)\")\nplt.plot(t, beta_wave, label=\"Beta (12-30 Hz)\")\nplt.plot(t, gamma_wave, label=\"Gamma (30+ Hz)\")\n\nplt.xlabel(\"Time (s)\")\nplt.ylabel(\"Amplitude\")\nplt.title(\"Brain Wave Frequencies\")\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:39.190024Z","iopub.execute_input":"2025-02-26T23:46:39.190421Z","iopub.status.idle":"2025-02-26T23:46:39.556536Z","shell.execute_reply.started":"2025-02-26T23:46:39.190382Z","shell.execute_reply":"2025-02-26T23:46:39.555519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport pandas as pd\n\nwaves = [\"Delta\", \"Theta\", \"Alpha\", \"Beta\", \"Gamma\"]\nfrequencies = [4, 8, 12, 30, 40]\ndf = pd.DataFrame({\"Brain Waves\": waves, \"Max Frequency (Hz)\": frequencies})\nplt.figure(figsize=(8,5))\nsns.barplot(x=\"Brain Waves\", y=\"Max Frequency (Hz)\", data=df, palette=\"coolwarm\")\nplt.title(\"Brain Wave Frequency Ranges\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:39.558409Z","iopub.execute_input":"2025-02-26T23:46:39.558774Z","iopub.status.idle":"2025-02-26T23:46:39.778389Z","shell.execute_reply.started":"2025-02-26T23:46:39.558742Z","shell.execute_reply":"2025-02-26T23:46:39.777338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.signal as signal\neeg_signal = delta_wave + theta_wave + alpha_wave + beta_wave + gamma_wave\nf, t, Sxx = signal.spectrogram(eeg_signal, fs=500)\nplt.figure(figsize=(10,6))\nplt.pcolormesh(t, f, Sxx, shading='gouraud')\nplt.ylabel('Frequency (Hz)')\nplt.xlabel('Time (s)')\nplt.title('Spectrogram of Simulated EEG Waves')\nplt.colorbar(label='Power')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:39.779776Z","iopub.execute_input":"2025-02-26T23:46:39.780161Z","iopub.status.idle":"2025-02-26T23:46:40.098503Z","shell.execute_reply.started":"2025-02-26T23:46:39.780121Z","shell.execute_reply":"2025-02-26T23:46:40.097242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset Challenges\n1️⃣ Complex Data Paths – Multiple file paths for EEG and spectrograms make retrieval and management difficult.\n\n2️⃣ EEG vs. Spectrogram Differences – EEG provides raw time-series data, while spectrograms offer frequency insights but may lose temporal precision.\n\n3️⃣ Class Imbalance – Uneven distribution of labels can bias model predictions, requiring balancing techniques.\n\n4️⃣ Label Uncertainty – Expert disagreements lead to ambiguous classifications.\n\n5️⃣ High Dimensionality – Large EEG datasets need feature extraction and dimensionality reduction.\n\n6️⃣ Noise & Artifacts – EEG signals contain artifacts (e.g., muscle movements) requiring preprocessing (e.g., filtering, ICA).","metadata":{}},{"cell_type":"markdown","source":"# 1. Data Import & Preprocessing","metadata":{"_uuid":"928a33ae-0842-4beb-87fb-c68660f621f0","_cell_guid":"bfef6318-404c-4df4-9297-b65a4050955a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import pandas as pd\nimport glob\nfrom tqdm import tqdm\nimport polars as pl\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt \nimport seaborn as sns","metadata":{"_uuid":"a0dd09e5-b5d9-4095-af59-68f8ebfcbbdc","_cell_guid":"aab8f356-6409-4f08-9934-8a5adea82eb4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.099620Z","iopub.execute_input":"2025-02-26T23:46:40.099959Z","iopub.status.idle":"2025-02-26T23:46:40.104969Z","shell.execute_reply.started":"2025-02-26T23:46:40.099921Z","shell.execute_reply":"2025-02-26T23:46:40.103747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"_uuid":"7f0fc089-4e3f-48f8-ae2c-3ff9eebf827a","_cell_guid":"3c2b38bd-30fd-463f-88bf-b27444fcad8f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.106074Z","iopub.execute_input":"2025-02-26T23:46:40.106409Z","iopub.status.idle":"2025-02-26T23:46:40.289369Z","shell.execute_reply.started":"2025-02-26T23:46:40.106370Z","shell.execute_reply":"2025-02-26T23:46:40.288486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define path and variables\n","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:40.290274Z","iopub.execute_input":"2025-02-26T23:46:40.290645Z","iopub.status.idle":"2025-02-26T23:46:40.295601Z","shell.execute_reply.started":"2025-02-26T23:46:40.290610Z","shell.execute_reply":"2025-02-26T23:46:40.294431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class config:\n    BATCH_SIZE = 64\n    FOLDS = 0\n    MODEL = \"tf_efficientnet_b0\"\n    SEED = 29\n\n\nclass_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\nlabel2name = dict(enumerate(class_names))\nname2label = {v:k for k, v in label2name.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:40.298441Z","iopub.execute_input":"2025-02-26T23:46:40.298885Z","iopub.status.idle":"2025-02-26T23:46:40.315266Z","shell.execute_reply.started":"2025-02-26T23:46:40.298850Z","shell.execute_reply":"2025-02-26T23:46:40.314149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Collect all tracks at once","metadata":{"_uuid":"cdc2cbb2-00be-404a-9068-c318651316ae","_cell_guid":"dc2ac5d4-1dee-4fa4-a70a-5676678b6f0c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"train_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*\")\ntrain_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*\")","metadata":{"_uuid":"b4f366f3-d29a-4262-a0b1-d592f7362300","_cell_guid":"121bda4d-be82-4840-a4e8-a43e14efc161","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.317162Z","iopub.execute_input":"2025-02-26T23:46:40.317449Z","iopub.status.idle":"2025-02-26T23:46:40.395612Z","shell.execute_reply.started":"2025-02-26T23:46:40.317425Z","shell.execute_reply":"2025-02-26T23:46:40.394749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Convert lists to dictionaries to improve searching.","metadata":{"_uuid":"cb6634b2-8d95-413c-99d3-491909e00cf5","_cell_guid":"f9742027-de2e-42c4-a603-d3fc280f9333","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"train_eeg_dict = {path.split(\"/\")[-1].split(\".\")[0]: path for path in train_eeg_path_list}\ntrain_spectrogram_dict = {path.split(\"/\")[-1].split(\".\")[0]: path for path in train_spectrograms_path_list}","metadata":{"_uuid":"ae83ef3f-50a8-424f-ae42-2abee9761fb8","_cell_guid":"981d7693-ec6c-4da1-9029-c3bec374288e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.396697Z","iopub.execute_input":"2025-02-26T23:46:40.397061Z","iopub.status.idle":"2025-02-26T23:46:40.421583Z","shell.execute_reply.started":"2025-02-26T23:46:40.397025Z","shell.execute_reply":"2025-02-26T23:46:40.420513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Link each EEG and spectrogram to its path using map","metadata":{"_uuid":"79978d61-6316-421f-8d6c-99de48085d82","_cell_guid":"b44e6ee0-141f-403c-9a9f-f93f92b92e62","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"train_df['eeg_path'] = train_df['eeg_id'].astype(str).map(train_eeg_dict)\ntrain_df['spectrograms_path'] = train_df['spectrogram_id'].astype(str).map(train_spectrogram_dict)","metadata":{"_uuid":"eeb5c43f-3afd-41ea-96e7-bc45b1e1b98d","_cell_guid":"6a6f08a0-4569-4a12-9c20-0b2584f8d065","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.422620Z","iopub.execute_input":"2025-02-26T23:46:40.422967Z","iopub.status.idle":"2025-02-26T23:46:40.545787Z","shell.execute_reply.started":"2025-02-26T23:46:40.422923Z","shell.execute_reply":"2025-02-26T23:46:40.544714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"_uuid":"c9e1956a-357d-4866-9a60-d83d7ca83587","_cell_guid":"d9860a87-ac47-4b02-9281-bcc527afe71b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.546807Z","iopub.execute_input":"2025-02-26T23:46:40.547097Z","iopub.status.idle":"2025-02-26T23:46:40.564935Z","shell.execute_reply.started":"2025-02-26T23:46:40.547072Z","shell.execute_reply":"2025-02-26T23:46:40.563549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")","metadata":{"_uuid":"80b3a963-5885-49a9-b339-1cf967d10dcd","_cell_guid":"e8b795c9-ee5f-44f5-a610-bc79afbaff49","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.566351Z","iopub.execute_input":"2025-02-26T23:46:40.566788Z","iopub.status.idle":"2025-02-26T23:46:40.581030Z","shell.execute_reply.started":"2025-02-26T23:46:40.566749Z","shell.execute_reply":"2025-02-26T23:46:40.579937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntest_eeg_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*\")\ntest_spectrograms_path_list = glob.glob(\"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*\")","metadata":{"_uuid":"58be4627-8fc3-4c1b-8995-d65fc362f5c7","_cell_guid":"7dee91c7-6eb5-41cd-9e50-66c97257cace","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.582021Z","iopub.execute_input":"2025-02-26T23:46:40.582387Z","iopub.status.idle":"2025-02-26T23:46:40.588537Z","shell.execute_reply.started":"2025-02-26T23:46:40.582344Z","shell.execute_reply":"2025-02-26T23:46:40.587635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntest_eeg_dict = {path.split(\"/\")[-1].split(\".\")[0]: path for path in test_eeg_path_list}\ntest_spectrogram_dict = {path.split(\"/\")[-1].split(\".\")[0]: path for path in test_spectrograms_path_list}","metadata":{"_uuid":"cc642903-1d2c-4294-b28b-c4113d215d39","_cell_guid":"80b0e70c-91b0-4981-948d-336a7539e87d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.589573Z","iopub.execute_input":"2025-02-26T23:46:40.589876Z","iopub.status.idle":"2025-02-26T23:46:40.605964Z","shell.execute_reply.started":"2025-02-26T23:46:40.589852Z","shell.execute_reply":"2025-02-26T23:46:40.605015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntest_df['eeg_path'] = test_df['eeg_id'].astype(str).map(test_eeg_dict)\ntest_df['spectrograms_path'] = test_df['spectrogram_id'].astype(str).map(test_spectrogram_dict)","metadata":{"_uuid":"f0eff6a6-3cb1-4da4-aaaa-caf91d549a7d","_cell_guid":"6f408604-4e77-474a-84a9-89cd24edad82","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-02-26T23:46:40.606979Z","iopub.execute_input":"2025-02-26T23:46:40.607346Z","iopub.status.idle":"2025-02-26T23:46:40.627282Z","shell.execute_reply.started":"2025-02-26T23:46:40.607310Z","shell.execute_reply":"2025-02-26T23:46:40.626377Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"_uuid":"af0cea08-22ac-42cd-8a53-9cc2d1255694","_cell_guid":"19fefca1-698b-4728-ba6f-74ae1b83facc","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-02-26T23:46:40.628306Z","iopub.execute_input":"2025-02-26T23:46:40.628618Z","iopub.status.idle":"2025-02-26T23:46:40.648531Z","shell.execute_reply.started":"2025-02-26T23:46:40.628584Z","shell.execute_reply":"2025-02-26T23:46:40.647487Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**How are distributed the train data.**","metadata":{}},{"cell_type":"code","source":"df_train = pl.read_csv(f'{BASE_PATH}/train.csv')\ndf_train = df_train.with_columns(\n    eeg_path = f'{BASE_PATH}/train_eegs/' + pl.col('eeg_id').cast(pl.Utf8) + '.parquet',\n    spec_path = f'{BASE_PATH}/train_spectrograms/' + pl.col('spectrogram_id').cast(pl.Utf8) + '.parquet',\n    class_label = pl.col('expert_consensus').replace(name2label).str.to_integer()\n)\ndf_train.head(9)","metadata":{"_uuid":"74d4bb28-5285-48b3-8c4f-8f150a402b61","_cell_guid":"28270baf-8cb4-4b6d-b70f-f0324a9b1477","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.649629Z","iopub.execute_input":"2025-02-26T23:46:40.650014Z","iopub.status.idle":"2025-02-26T23:46:40.713322Z","shell.execute_reply.started":"2025-02-26T23:46:40.649975Z","shell.execute_reply":"2025-02-26T23:46:40.712092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### compare the total of registers with the number of IDs of each identifier we have (EEG, Spectogram, and patient)","metadata":{}},{"cell_type":"code","source":"df_train.shape[0]","metadata":{"_uuid":"b97348fa-3ec0-4740-ba3e-a5b7149be2bd","_cell_guid":"961a677e-a35b-4ded-83de-9c275c986f05","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.714531Z","iopub.execute_input":"2025-02-26T23:46:40.714826Z","iopub.status.idle":"2025-02-26T23:46:40.721108Z","shell.execute_reply.started":"2025-02-26T23:46:40.714793Z","shell.execute_reply":"2025-02-26T23:46:40.719969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.group_by('eeg_id').len().shape[0]","metadata":{"_uuid":"7c699a06-1282-44f6-9a8b-23ee993ab1a0","_cell_guid":"df3b7cb5-5ba8-4178-b6a9-423e1b5c84be","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.722308Z","iopub.execute_input":"2025-02-26T23:46:40.722684Z","iopub.status.idle":"2025-02-26T23:46:40.754963Z","shell.execute_reply.started":"2025-02-26T23:46:40.722646Z","shell.execute_reply":"2025-02-26T23:46:40.753968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.group_by('spectrogram_id').len().shape[0]","metadata":{"_uuid":"519667de-0535-4fbb-8e2d-41fe46d7d297","_cell_guid":"573a7589-a85b-45df-abfe-60b6c4eb9ff2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.756017Z","iopub.execute_input":"2025-02-26T23:46:40.756388Z","iopub.status.idle":"2025-02-26T23:46:40.776159Z","shell.execute_reply.started":"2025-02-26T23:46:40.756350Z","shell.execute_reply":"2025-02-26T23:46:40.775113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.group_by('patient_id').len().shape[0]","metadata":{"_uuid":"85439d5f-9762-49c2-a2ed-005261f4f473","_cell_guid":"68f67639-2e6d-4404-a8fd-d7a548db47e8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-02-26T23:46:40.777153Z","iopub.execute_input":"2025-02-26T23:46:40.777505Z","iopub.status.idle":"2025-02-26T23:46:40.800882Z","shell.execute_reply.started":"2025-02-26T23:46:40.777462Z","shell.execute_reply":"2025-02-26T23:46:40.799831Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**As it can be seem, there is quite less IDs in the dataset compare to the total registers. So, we can assume there exist a kind of overlapping between the ids**","metadata":{}},{"cell_type":"markdown","source":"## Expert consensus","metadata":{}},{"cell_type":"markdown","source":"The expert consensus is picked by the column (seizure_vote, lpd_vote, gpd_vote, lrda_vote, grda_vote, other_vote) with most votes.","metadata":{}},{"cell_type":"code","source":"import warnings\n\nwarnings.simplefilter(action='ignore', category=FutureWarning)\nwarnings.simplefilter(action='ignore', category=UserWarning)\n\nhisto = df_train.group_by('expert_consensus').len().sort('len', descending=True).to_pandas()\nplt.grid(True, linestyle='--', color='gray', linewidth=0.5, alpha=0.3)\nsns.histplot(data=histo,\n             x='expert_consensus',\n             weights='len',\n             discrete=True,\n             shrink=0.8,\n             hue='expert_consensus',\n             palette='pastel',\n             legend=False)\n\nplt.title('Histogram of expert consensus')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:40.804964Z","iopub.execute_input":"2025-02-26T23:46:40.805288Z","iopub.status.idle":"2025-02-26T23:46:41.223500Z","shell.execute_reply.started":"2025-02-26T23:46:40.805257Z","shell.execute_reply":"2025-02-26T23:46:41.222406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.heatmap(df_train.select('seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote').corr(),\n            annot=True,\n            cmap='coolwarm',\n            fmt=\".2f\",\n            linewidths=0.5,\n            xticklabels=['seizure','lpd','gpd','lrda','grda','other'],\n            yticklabels=['seizure','lpd','gpd','lrda','grda','other'])\nplt.title('Correlation Matrix between experts\\' votes')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.225227Z","iopub.execute_input":"2025-02-26T23:46:41.225707Z","iopub.status.idle":"2025-02-26T23:46:41.560858Z","shell.execute_reply.started":"2025-02-26T23:46:41.225661Z","shell.execute_reply":"2025-02-26T23:46:41.559756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Probability of class","metadata":{}},{"cell_type":"markdown","source":"> The primary goal is to understand how the \"expert_consensus\" labels are distributed in the dataset under different conditions.","metadata":{}},{"cell_type":"code","source":"target_votes = df_train.group_by('expert_consensus').len()\ntotal_votes = target_votes['len'].sum()\nmean_votes = target_votes.with_columns(pl.col('len')/total_votes)\n\nmean_votes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.561964Z","iopub.execute_input":"2025-02-26T23:46:41.562263Z","iopub.status.idle":"2025-02-26T23:46:41.573327Z","shell.execute_reply.started":"2025-02-26T23:46:41.562222Z","shell.execute_reply":"2025-02-26T23:46:41.572271Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":">  This provides a baseline understanding of how frequently each EEG class appears in the dataset.","metadata":{}},{"cell_type":"code","source":"lol = df_train.group_by('eeg_id').first()\ntarget_votes = lol.group_by('expert_consensus').len()\ntotal_votes = target_votes['len'].sum()\nmean_votes = target_votes.with_columns(pl.col('len')/total_votes)\n\nmean_votes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.574562Z","iopub.execute_input":"2025-02-26T23:46:41.575033Z","iopub.status.idle":"2025-02-26T23:46:41.607508Z","shell.execute_reply.started":"2025-02-26T23:46:41.574992Z","shell.execute_reply":"2025-02-26T23:46:41.606554Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> This checks if EEG classification changes when considering only unique EEG recordings rather than all samples.","metadata":{}},{"cell_type":"code","source":"lol = df_train.group_by('spectrogram_id').first()\ntarget_votes = lol.group_by('expert_consensus').len()\ntotal_votes = target_votes['len'].sum()\nmean_votes = target_votes.with_columns(pl.col('len')/total_votes)\n\nmean_votes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.608530Z","iopub.execute_input":"2025-02-26T23:46:41.608859Z","iopub.status.idle":"2025-02-26T23:46:41.635686Z","shell.execute_reply.started":"2025-02-26T23:46:41.608831Z","shell.execute_reply":"2025-02-26T23:46:41.634607Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> This checks if the spectrogram representation of EEG signals affects classification probabilities.","metadata":{}},{"cell_type":"code","source":"lol = df_train.group_by('patient_id').first()\ntarget_votes = lol.group_by('expert_consensus').len()\ntotal_votes = target_votes['len'].sum()\nmean_votes = target_votes.with_columns(pl.col('len')/total_votes)\n\nmean_votes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.636684Z","iopub.execute_input":"2025-02-26T23:46:41.636962Z","iopub.status.idle":"2025-02-26T23:46:41.657537Z","shell.execute_reply.started":"2025-02-26T23:46:41.636938Z","shell.execute_reply":"2025-02-26T23:46:41.656594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> This checks if the spectrogram representation of EEG signals affects classification probabilities.","metadata":{}},{"cell_type":"markdown","source":"## EEG X SPEC\n\n##### EEG Signal: Represents the electrical activity of the brain over time.\n##### Spectrogram: Shows how the signal's frequency components change over time, useful for identifying brainwave patterns (Delta, Theta, Alpha, Beta waves).","metadata":{}},{"cell_type":"code","source":"pqf_spec = pl.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')\npqf_spec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.658639Z","iopub.execute_input":"2025-02-26T23:46:41.659002Z","iopub.status.idle":"2025-02-26T23:46:41.682841Z","shell.execute_reply.started":"2025-02-26T23:46:41.658968Z","shell.execute_reply":"2025-02-26T23:46:41.681766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n\npqf_eeg = pl.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1002576868.parquet')\npqf_eeg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-26T23:46:41.683815Z","iopub.execute_input":"2025-02-26T23:46:41.684123Z","iopub.status.idle":"2025-02-26T23:46:41.696873Z","shell.execute_reply.started":"2025-02-26T23:46:41.684097Z","shell.execute_reply":"2025-02-26T23:46:41.695731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}