{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7438040,"sourceType":"datasetVersion","datasetId":4329040},{"sourceId":7457180,"sourceType":"datasetVersion","datasetId":4313900},{"sourceId":7526248,"sourceType":"datasetVersion","datasetId":4308295},{"sourceId":6130,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n---\n\n**Hi! I'm giving up on this competition (ended up way too busy with work and I want to move on to something else). That said I wanted to make it public. Use it if you'd like and I hope it helps a little in some way!**\n\n---\n\n<br>","metadata":{}},{"cell_type":"code","source":"# Run this to enable CSS types\nfrom IPython.core.display import HTML\ndef css_styling():\n    styles = open(\"/kaggle/input/my-css-styles/kaggle_styles.css\", \"r\").read()\n    return HTML(\"<style>\"+styles+\"</style>\")\ncss_styling()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-01-28T18:25:18.593505Z","iopub.execute_input":"2024-01-28T18:25:18.594012Z","iopub.status.idle":"2024-01-28T18:25:18.611049Z","shell.execute_reply.started":"2024-01-28T18:25:18.593986Z","shell.execute_reply":"2024-01-28T18:25:18.610187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n<div style=\"background-color: #1B5264; padding: 25px 0; text-align: center;\">\n    <img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/hms_comp_banner.png?raw=true\" alt=\"Notebook Header Image\" style=\"padding-top: 25px 0 0 0;\">\n    <h2 style=\"font-family: Verdana; font-size: 32px; font-weight: bold; color: #A9E8DC; font-variant: small-caps; letter-spacing: 2px;\">        \n        Classify seizures and other patterns of harmful brain activity in critically ill patients\n    </h2><br>\n    <span style=\"font-size: 20px; color: white; letter-spacing: 2px; font-weight: bold;\">📚LEARN📚 &nbsp;&nbsp;&nbsp; 🔭EDA🔭 &nbsp;&nbsp;&nbsp; 🤖BASELINE🤖</span>\n    <h5 style=\"font-family: Verdana; font-size: 12px; font-weight: bold; color: white;\">\n        CREATED BY: DARIEN SCHETTLER\n    </h5>\n</div>\n\n<hr>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n<hr>","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #1B5264; background-color: #ffffff;\">\n    TABLE OF CONTENTS\n</h1>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#introduction\" style=\"text-decoration: none; color: #77AAB0;\">1&nbsp;&nbsp;&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#background_information\" style=\"text-decoration: none; color: #77AAB0;\">2&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#imports\" style=\"text-decoration: none; color: #77AAB0;\">3&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#setup\" style=\"text-decoration: none; color: #77AAB0;\">4&nbsp;&nbsp;&nbsp;&nbsp;SETUP & HELPER FUNCTIONS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#eda\" style=\"text-decoration: none; color: #77AAB0;\">5&nbsp;&nbsp;&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#baseline\" style=\"text-decoration: none; color: #77AAB0;\">6&nbsp;&nbsp;&nbsp;&nbsp;BASELINE</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#next_steps\" style=\"text-decoration: none; color: #77AAB0;\">7&nbsp;&nbsp;&nbsp;&nbsp;NEXT STEPS</a></h3>\n\n<hr>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"introduction\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"introduction\">1&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">1.1 <b>WHAT</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li style=\"font-family: Verdana;\">This notebook will follow the authors learning path and highlight relevant terms, information, and useful content about the competition.</li>\n    <li>This notebook will conduct an <b>E</b>xploratory <b>D</b>ata <b>A</b>nalysis for the competition.</li>\n    <li>This notebook will propose an open-source baseline solution.</li>\n</ul>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">1.2 <b>WHY</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li>Writing and sharing my learning path and the resulting exploratory data analysis can help improve my own understanding of the competition and the data.</li>\n    <li>Sharing my work may help others who are interested in the competition (or the data). This help may take the form of:\n        <ul>\n            <li>Better understanding the problem and potential common solutions (incl. my baseline).</li>\n            <li>Better understanding of the provided dataset.</li>\n            <li>Better understanding of the background information and research.</li>\n            <li>Better ability to hypothesize new solutions.</li>\n        </ul>\n    </li>\n    <li>Exploratory data analysis is a critical step in any data science project. Sharing my EDA might help others in the competition.</li>\n    <li>Writing and sharing my work is often a fun and rewarding experience! It not only allows me to explore and try different techniques, ideas, and visualizations but also encourages and supports other learners and participants.</li>\n</ul>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">1.3 <b>WHO</b> IS THIS FOR?</h3>\n<hr>\n\n<ul>\n    <li>The primary purpose of this notebook is to educate <b>MYSELF</b>, however, my review/learning might be beneficial to others:\n        <ul>\n            <li>Other Kagglers (aka. current and future competition participants).</li>\n            <li>Anyone interested in learning more about sign language recognition and its potential applications.</li>\n            <li>Educators, students, or researchers who want to gain hands-on experience working with real-world data and building machine learning models and want to follow along with something relatively straightforward.</li>\n            <li>Those who want to learn how to use specific tools (competition specific and data science) and libraries such as TensorFlow Lite, MediaPipe, pandas, numpy, etc.</li>\n        </ul>\n    </li>\n</ul>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">1.4 <b>HOW</b> WILL THIS WORK?</h3>\n<hr>\n\n<p>I'm going to assemble some markdown cells (like this one) at the beginning of the notebook to go over some concepts/details/etc.</p>\n\n<p>Following this, I will attempt to walk through the data and understand it better prior to composing a baseline solution.</p>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"background_information\">2&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">2.1 <b>OVERVIEW</b></h3>\n<hr>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">PRIMARY TASK DESCRIPTION</b>\n    <br>\n    <br>\n    <div style=\"font-family: Verdana !important;\">\n        The goal of this competition is to <b>detect and classify seizures and other types of harmful brain activity</b>.\n        <br>\n        You will develop a <b>model trained on electroencephalography (EEG) signals</b> recorded from hospital patients.\n        <br>\n        <br>\n        <b>The evaluation metric for this contest is <mark><a style=\"color: black !important;\" href=\"https://www.wikiwand.com/en/Kullback%E2%80%93Leibler_divergence\">Kullback Liebler divergence</a></mark> between the predicted probability and the observed target</b>\n    </div>\n</div>\n\n<br>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">ELI5 TASK DESCRIPTION</b>\n    <br>\n    <sup><b>ChatGPT was used to help generate this Analogy</b></sup>\n    <br>\n    <br>\n    <div style=\"font-family: Verdana;\">\n        Imagine your brain is like a <b>busy city with lots of traffic – the cars are like the electrical signals in your brain</b>.\n        <br>\n        <br>\n        Doctors use a special tool called an <span style=\"background-color: yellow;\"><b>EEG</b></span>, which is like a big map that shows them how the traffic (or brain signals) is moving.\n        <br>\n        This helps them see if there's anything unusual, like a traffic jam (which could be a seizure or other problems in the brain).\n        <br>\n        <br>\n        In this competition, people are trying to make a computer program that can look at the EEG maps and quickly find out if there's a traffic jam or any other problems, without needing a doctor to look at it every time.\n        <br>\n        <br>\n        This is important because it can help patients really fast and also make sure that doctors don't get too overwhelmed looking at so many 'maps'.\n    </div>\n    <div style=\"font-family: Verdana !important;\">\n        <br>\n        The competition focuses on <b>six types of brain</b> signal patterns:\n        <ul>\n            <li><span style=\"font-weight: bold;\">Seizure (SZ):</span> Like a big, sudden traffic jam.</li>\n            <li><span style=\"font-weight: bold;\">Generalized Periodic Discharges (GPD):</span> Like regular stops in traffic at certain spots.</li>\n            <li><span style=\"font-weight: bold;\">Lateralized Periodic Discharges (LPD):</span> Similar to GPD, but only on one side of the brain.</li>\n            <li><span style=\"font-weight: bold;\">Lateralized Rhythmic Delta Activity (LRDA):</span> Slow, steady traffic on one side of the brain.</li>\n            <li><span style=\"font-weight: bold;\">Generalized Rhythmic Delta Activity (GRDA):</span> Slow, steady traffic all over the brain.</li>\n            <li><span style=\"font-weight: bold;\">\"Other\":</span> Anything that doesn't fit into the first five categories.</li>\n        </ul>\n    </div>\n    <div style=\"font-family: Verdana !important;\">\n        Sometimes, <b>the doctors (or experts) who look at the EEG maps don't agree on what they see.</b>\n        <br>\n        For example, they might not be sure if there's a traffic jam (seizure) or just slow traffic.\n        <br>\n        <br>\n        <b>In the competition, they have examples where:</b>\n        <ul style=\"font-family: Verdana;\" >\n            <li>Everyone agrees on what they see (like everyone agrees there's a traffic jam).</li>\n            <li>About half agree and half don't (like some think there's a traffic jam, and others think it's just slow traffic).</li>\n            <li>They are split between two different things (like some think it's a traffic jam, and others think it's a regular stop in traffic).</li>\n        </ul>\n    </div>\n    <div style=\"font-family: Verdana;\">\n        The goal is to make a computer program that can understand these different patterns, even when they are a bit confusing, so doctors can help patients faster and more accurately.\n    </div>\n</div>\n\n<br>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">COMPETITION HOST</b>\n    <br>\n    <ul>\n        <li><b>Sunstella Foundation:</b> Created in 2021 during the COVID pandemic to help minority graduate students in technology overcome challenges and celebrate their achievements.</li>\n        <li><b>Persyst, Jazz Pharmaceuticals, and the Clinical Data Animation Center (CDAC):</b> Research partners aiming to help preserve and enhance brain health.</li>\n    </ul>\n</div>\n\n<br>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST TASK DESCRIPTION</b>\n    <br>\n    <br>\n    <div style=\"font-family: Verdana;\">\n        From stethoscopes to tongue depressors, doctors rely on many tools to treat their patients. Physicians use electroencephalography with critically ill patients to detect seizures and other types of brain activity that can cause brain damage. \n        <br>\n        <br>\n        EEG monitoring currently relies solely on manual analysis by specialized neurologists. This labor-intensive process is time-consuming, expensive, and prone to fatigue-related errors and reliability issues between different reviewers.\n    </div>\n    <div style=\"font-family: Verdana;\">\n        Learn more about EEG signals interpretation in these videos: <b><a href=\"https://www.youtube.com/watch?v=S9NLrhj0x-M&t\">EEG Talk - ACNS Critical Care EEG Terminology 2021</a></b>\n        <ul>\n            <li><a href=\"https://www.youtube.com/watch?v=S9NLrhj0x-M&t\">Part 1</a></li>\n            <li><a href=\"https://www.youtube.com/watch?v=4D9R2WIKr-A%20\">Part 2</a></li>\n            <li><a href=\"https://www.youtube.com/watch?v=-R5yUX7p_j4%20\">Part 3</a></li>\n            <li><a href=\"https://www.youtube.com/watch?v=OknS2ObD9-g&t%20\">Part 4</a></li>\n            <li><a href=\"https://www.youtube.com/watch?v=2c7ABQRkn3s\">Part 5</a></li>\n        </ul>\n    </div>\n</div>\n\n<br>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HOST TASK DESCRIPTION DETAILS</b>\n    <br>\n    <br>\n    <div style=\"font-family: Verdana !important;\">\n        Your work in automating EEG analysis will help detect seizures and other brain activities, enabling faster and more accurate treatments. The algorithms developed may also aid in drug development to treat and prevent seizures.\n        <br>\n        <br>\n        There are <b>six</b> patterns of interest for this competition:\n        <ul>\n            <li>Seizure (SZ)</li>\n            <li>Generalized periodic discharges (GPD)</li>\n            <li>Lateralized periodic discharges (LPD)</li>\n            <li>Lateralized rhythmic delta activity (LRDA)</li>\n            <li>Generalized rhythmic delta activity (GRDA)</li>\n            <li>“Other”</li>\n        </ul>\n    </div>\n    <div style=\"font-family: Verdana !important;\">\n        The EEG segments used in this competition have been annotated by experts, including:\n        <ul>\n            <li><b>'Idealized'</b> patterns with high agreement</li>\n            <li><b>'Proto patterns'</b> with mixed opinions</li>\n            <li><b>'Edge cases'</b> with split decisions between two patterns</li>\n        </ul>\n        Detailed explanations of these patterns are <a href=\"https://www.acns.org/UserFiles/file/ACNSStandardizedCriticalCareEEGTerminology_rev2021.pdf\">available here</a>\n    </div>\n</div>\n    \n<br>\n\n<div style=\"font-family: Verdana !important;\">\n    <br>\n    <div style=\"font-family: Verdana !important;\">\n        <b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">EXAMPLES OF EEG PATTERNS</b>\n        <br>\n        <br>\n        <a href=\"https://storage.googleapis.com/kaggle-media/competitions/Harvard%20Medical%20School/eFig2.png\" target=\"_blank\"><img width=\"80%\" alt=\"EEG Patterns\" src=\"https://storage.googleapis.com/kaggle-media/competitions/Harvard%20Medical%20School/eFig2.png\"></a>\n    </div>\n    <div style=\"font-family: Verdana; font-size:8px; font-weight: bold;\">    \n        Please refer to <a href=\"https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/data\">Data tab</a> for full screen PDF page of each subfigure.\n        <br>\n        <br>\n        This figure shows selected examples of EEG patterns with different level of agreement. Rows are structured with the 1st row seizure, 2nd row LPDs, 3rd row GPDs, 4th row LRDA, and 5th row GRDA. Column-wise, examples of idealized forms of patterns are in the 1st column (A). These are patterns with uniform expert agreement. The 2nd column (B) are proto or partially formed patterns. About half of raters labeled these as one IIIC pattern and the other half labeled “Other”. The 3rd and 4th columns (C, D) are edge cases (about half of raters labeled these one IIIC pattern and half labeled them as another IIIC pattern).\n        <br>\n        <br>\n        For B-1 there is rhythmic delta activity with some admixed sharp discharges within the 10 second raw EEG, and the spectrogram shows that this segment may belong to the tail end of a seizure, thus disagreement between SZ and “Other” makes sense. B-2 shows frontal lateralized sharp transients at ~1Hz, but they have a reversed polarity, suggesting they may be coming from a non-cerebral source, thus the split between LPD and “Other” (artifact) makes sense. B-3 has diffused semi-rhythmic delta background with poorly formed low amplitude generalized periodic discharges with s shifting morphology making it a proto-GPD type pattern. B-4 shows semi-rhythmic delta activity with unstable morphology over the right hemisphere, a proto-LRDA pattern. B-5 shows a few waves of rhythmic delta activity with an unstable morphology and is poorly sustained, a proto-GRDA. C-1 shows 2Hz LPDs showing an evolution with increasing amplitude evolving underlying rhythmic activity, a pattern between LPDs and the beginning of a seizure, an edge-case. D-1 shows abundant GPDs on top of a suppressed background with frequency of 1-2Hz. The average over the 10-seconds is close to 1.5Hz, suggesting a seizure, another edge case. C-2 is split between LPDs and GPDs. The amplitude of the periodic discharges is higher over the right, but a reflection is also seen on the left. D-2 is tied between LPDs and LRDA. It shares some features of both; in the temporal derivations it looks more rhythmic whereas in the parasagittal derivations it looks periodic. C-3 is split between GPDs and LRDA. The ascending limb of the delta waves have a sharp morphology, and these periodic discharges are seen on both sides. The rhythmic delta appears to be of higher amplitude over the left, but there is some reflection of the activity on the left. D-3 is split between GPDs and GRDA. The ascending limb of the delta wave has a sharp morphology and there is asymmetry in slope between ascending and descending limbs making it an edge case. C-4 is split between LRDA and seizure. It shows 2Hz LRDA on the left, and the spectrogram shows that this segment may belong to the tail end of a seizure, an edge-case. D-4 is split between LRDA and GRDA. The rhythmic delta appears to be of higher amplitude over the left, but there is some reflection of the activity on the right. C-5 is split between GRDA and seizure. It shows potentially evolving rhythmic delta activity with poorly formed embedded epileptiform discharges, a pattern between GRDA and seizure, an edge-case. D-5 is split between GRDA and LPDs. There is generalized rhythmic delta activity, while the activity on the right is somewhat higher amplitude and contains poorly formed epileptiform discharges suggestive of LPDs, an edge-case. Note: Recording regions of the EEG electrodes are abbreviated as LL = left lateral; RL = right lateral; LP = left parasagittal; RP = right parasagittal.\n    </div>\n</div>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">2.2 <b>DATASET INFORMATION</b></h3>\n\n<hr>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">HIGH LEVEL DATA SUMMARY</b>\n\nThis dataset is structured to aid in detecting and classifying seizures and other types of harmful brain activity using EEG (electroencephalography) data.<br>\n\n**DATA COMPOSITION**\n- **EEG Recordings:** Unique EEG recordings, each identified by `eeg_id`.\n- **Subsamples:** 50-second EEG subsamples (`eeg_sub_id`) and 10-minute spectrogram subsamples (`spectrogram_sub_id`).\n- **Annotator Votes:** Data includes counts of votes from annotators for various brain activity classes like seizures, LPD (lateralized periodic discharges), GPD (generalized periodic discharges), LRDA (lateralized rhythmic delta activity), GRDA (generalized rhythmic delta activity), and others.\n- **Patient Information:** Each record is linked to a specific patient (`patient_id`).\n\n**DATA FORMAT AND FILES**\n- **CSV Metadata Files:** `train.csv` and `test.csv` containing metadata like IDs, offsets, and votes.\n- **EEG Data Folders:** Separate folders for train (`train_eegs/`) and test (`test_eegs/`) containing EEG readings.\n- **Spectrogram Folders:** Similar structure as EEG data, with folders for train (`train_spectrograms/`) and test (`test_spectrograms/`) spectrograms.\n- **Example Images:** `example_figures/` folder contains larger copies of example images.\n\n**UNIQUE ASPECTS OF THE DATASET**\n- **Overlapping Samples:** Training data consists of overlapping samples, making it complex and closer to real-world scenarios.\n- **Expert Consensus:** The dataset includes an `expert_consensus` field, providing a reference label based on majority voting by experts.\n- **Spectrogram Data:** In addition to raw EEG data, the dataset includes spectrograms, offering a different perspective on the EEG signals.\n\n**CHALLENGES AND OPPORTUNITIES**\n- **Complexity in Interpretation:** The variation in expert opinions reflects the complexity of interpreting EEG data.\n- **Multi-faceted Analysis:** The dataset allows for both time-domain (EEG) and frequency-domain (spectrogram) analysis.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">DATA FILE DESCRIPTIONS</b>\n\n<b><code>train.csv</code> Metadata for the train set:</b>\n* <b><code>eeg_id</code> (int)</b>\n    * A unique identifier for the entire EEG recording.\n* <b><code>eeg_sub_id</code> (int)</b>\n    * An ID for the specific 50-second long subsample this row's labels apply to.\n* <b><code>eeg_label_offset_seconds</code> (int)</b>\n    * The time between the beginning of the consolidated EEG and this subsample.\n* <b><code>spectrogram_id</code> (int)</b>\n    * A unique identifier for the entire EEG recording.\n* <b><code>spectrogram_sub_id</code> (int)</b>\n    * An ID for the specific 10-minute subsample this row's labels apply to.\n* <b><code>spectogram_label_offset_seconds</code> (int)</b>\n    * The time between the beginning of the consolidated spectrogram and this subsample.\n* <b><code>label_id</code> (int)</b>\n    * An ID for this set of labels.\n* <b><code>patient_id</code> (int)</b>\n    * An ID for the patient who donated the data.\n* <b><code>expert_consensus</code> (string)</b>\n    * The consensus annotator label. Provided for convenience only.\n* <b>Voting columns (<code>[seizure/lpd/gpd/lrda/grda/other]_vote</code>)</b>\n    * The count of annotator votes for a given brain activity class.\n\n<b><code>test.csv</code> Metadata for the test set:</b>\n* Similar to `train.csv`, but many columns don't apply due to no overlapping samples in the test set.\n\n<b><code>sample_submission.csv</code> format:</b>\n* Requires predictions to be probabilities for each class.\n* The target columns are similar to the voting columns in `train.csv`.\n\n<b><code>[train|test]_eegs/</code> folders:</b>\n* Contain EEG data with columns named after individual electrode locations and an EKG column.\n* Data collected at a frequency of 200 samples per second.\n\n<b><code>[train|test]_spectrograms/</code> folders:</b>\n* Spectrograms assembled from EEG data.\n* Column names indicate frequency in hertz and recording regions of the EEG electrodes.\n\n<b><code>example_figures/</code> folder:</b>\n* Contains larger copies of the example case images used on the overview tab.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">POST-EDA DATA OBSERVATIONS</b>\n\nTBD\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">2.3 <b>EVALUATION INFORMATION</b></h3>\n<hr>\n\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">BASIC EXPLANATION OF THE METRIC</b>\n\n<b>K</b>ullback-<b>L</b>eibler <b>(KL)</b> divergence, also known as <b>relative entropy</b>, is a measure used in statistical mathematics to <b>quantify how different one probability distribution is from another</b>. \n\nIt's a way to measure the <b>\"distance\"</b> between two distributions, but <mark>unlike familiar distances, it's <b>not symmetric</b> and <b>doesn't satisfy the triangle inequality</b></mark>. \n* <b>A KL divergence of 0 means the two distributions have the same information content</b>\n\nIt's widely used in various fields, including information theory, statistics, and machine learning, to compare distributions and model the information gain in different scenarios.\n\n<br>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">ELI5 EXPLANATION OF THE METRIC</b>\n\nImagine you have a <b>BIG jar of mixed <span style=\"font-weight: bold; color: red;\">c</span><span style=\"font-weight: bold; color: orange;\">a</span><span style=\"font-weight: bold; color: darkyellow;\">n</span><span style=\"font-weight: bold; color: green;\">d</span><span style=\"font-weight: bold; color: blue;\">i</span><span style=\"font-weight: bold; color: indigo;\">e</span><span style=\"font-weight: bold; color: violet;\">s</span></b>, and you're trying to guess what kinds are inside without looking. \n\nYou and your friend both make a guess. \n* You say, <i>\"I think it's <b>70% <span style=\"color: blue;\">blueberry sours</span></b> and <b>30% <span style=\"color: red;\">red vines</span></b>\"</i> \n* Your friend guesses, <i>\"It looks like <b>50% <span style=\"color: blue;\">blueberry sours</span></b> and <b>50% <span style=\"color: red;\">red vines</span></b>\"</i> \n\n<br>\n\n<u><b>Kullback-Leibler divergence</b></u> is like a special calculator that tells you how different your guesses are, but in a specific way. It measures how much extra information you would need if you used your friend's guess to figure out the actual mix in the jar based on your guess.\n\n**Think of it like this:**\n> If you were to bet on what kind of candy you'll pull out of the jar, KL divergence tells you **how much more likely you are to be surprised if you were betting based on your friend's guess instead of your own**.\n\nThe key thing to remember is that KL divergence is a one-way street. \n* It only tells you the 'surprise' from using your friend's guess instead of yours, <u>not the other way around</u>. \n* If you calculate it the other way (your friend's surprise using your guess), ***you might get a different number***.\n\nSo, KL divergence is all about comparing guesses (or predictions) and seeing how much one guess might lead you astray from another. \n\n**The bigger the number, the more one guess differs from the other in terms of expectations**.\n\n<br>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">EXPLANATION OF THE METRIC WITH RESPECT TO THIS COMPETITION</b>\n\nIn this competition, you are developing a model to classify EEG signals into several categories (like seizure, LPD, GPD, etc.). Our models will output probabilities for each category (classification), indicating how confident it is that the EEG signal belongs to each category.\n\nWhen we submit to the competition we are working with <b>TWO DISTINCT DISTRIBUTIONS:</b>\n* **The Reference Distribution**: \n    * This would be the 'true' classification of the EEG signals, which, ideally, is based on expert annotation.\n* **Our Model's Distribution**: \n    * The probabilities our model predicts for each category of EEG signals.\n\n<b>KL Divergence</b>, as explored above, is used here to evaluate how well our model's predictions (probability distributions) match the actual ground truth values (the reference distribution). \n* As explained above, the metric measures the difference between predicted probabilities (our models) and the true distribution (as indicated by expert annotations)\n\n<mark><b>In the Context of the Competition</b>, when our model predicts the probabilities for each EEG pattern, KL Divergence calculates how much your model's predictions deviate from the actual annotations.</mark>\n* The goal is to minimize this divergence. \n* A lower KL Divergence score means your model’s predictions are closer to the true distribution, indicating better performance.\n\n<br>\n\n<b>NOTE:</b>\n\nThe competition data includes cases where there's high agreement among experts as well as cases with less agreement. <b>This adds complexity because your model not only needs to be accurate in clear-cut cases (high agreement) <mark>but also needs to navigate the uncertainty in less clear cases (low agreement or split opinions).</mark></b>\n\n<br>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\"><a href=\"https://www.kaggle.com/code/metric/kullback-leibler-divergence/notebook\">ORIGINAL IMPLEMENTATION</a> AS PER KAGGLE</b>\n    \n```python\nimport numpy as np\nimport pandas as pd\nimport pandas.api.types\n\nimport kaggle_metric_utilities\n\nfrom typing import Optional\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\ndef kl_divergence(solution: pd.DataFrame, submission: pd.DataFrame, epsilon: float, micro_average: bool, sample_weights: Optional[pd.Series]):\n    # Overwrite solution for convenience\n    for col in solution.columns:\n        # Prevent issue with populating int columns with floats\n        if not pandas.api.types.is_float_dtype(solution[col]):\n            solution[col] = solution[col].astype(float)\n\n        # Clip both the min and max following Kaggle conventions for related metrics like log loss\n        # Clipping the max avoids cases where the loss would be infinite or undefined, clipping the min\n        # prevents users from playing games with the 20th decimal place of predictions.\n        submission[col] = np.clip(submission[col], epsilon, 1 - epsilon)\n\n        y_nonzero_indices = solution[col] != 0\n        solution[col] = solution[col].astype(float)\n        solution.loc[y_nonzero_indices, col] = solution.loc[y_nonzero_indices, col] * np.log(solution.loc[y_nonzero_indices, col] / submission.loc[y_nonzero_indices, col])\n        # Set the loss equal to zero where y_true equals zero following the scipy convention:\n        # https://docs.scipy.org/doc/scipy/reference/generated/scipy.special.rel_entr.html#scipy.special.rel_entr\n        solution.loc[~y_nonzero_indices, col] = 0\n\n    if micro_average:\n        return np.average(solution.sum(axis=1), weights=sample_weights)\n    else:\n        return np.average(solution.mean())\n\n\ndef score(\n        solution: pd.DataFrame,\n        submission: pd.DataFrame,\n        row_id_column_name: str,\n        epsilon: float=10**-15,\n        micro_average: bool=True,\n        sample_weights_column_name: Optional[str]=None\n    ) -> float:\n    ''' The Kullback–Leibler divergence.\n    The KL divergence is technically undefined/infinite where the target equals zero.\n\n    This implementation always assigns those cases a score of zero; effectively removing them from consideration.\n    The predictions in each row must add to one so any probability assigned to a case where y == 0 reduces\n    another prediction where y > 0, so crucially there is an important indirect effect.\n\n    https://en.wikipedia.org/wiki/Kullback%E2%80%93Leibler_divergence\n\n    solution: pd.DataFrame\n    submission: pd.DataFrame\n    epsilon: KL divergence is undefined for p=0 or p=1. If epsilon is not null, solution and submission probabilities are clipped to max(eps, min(1 - eps, p).\n    row_id_column_name: str\n    micro_average: bool. Row-wise average if True, column-wise average if False.\n\n    Examples\n    --------\n    >>> import pandas as pd\n    >>> row_id_column_name = \"id\"\n    >>> score(pd.DataFrame({'id': range(4), 'ham': [0, 1, 1, 0], 'spam': [1, 0, 0, 1]}), pd.DataFrame({'id': range(4), 'ham': [.1, .9, .8, .35], 'spam': [.9, .1, .2, .65]}), row_id_column_name=row_id_column_name)\n    0.216161...\n    >>> solution = pd.DataFrame({'id': range(3), 'ham': [0, 0.5, 0.5], 'spam': [0.1, 0.5, 0.5], 'other': [0.9, 0, 0]})\n    >>> submission = pd.DataFrame({'id': range(3), 'ham': [0, 0.5, 0.5], 'spam': [0.1, 0.5, 0.5], 'other': [0.9, 0, 0]})\n    >>> score(solution, submission, 'id')\n    0.0\n    >>> solution = pd.DataFrame({'id': range(3), 'ham': [0, 0.5, 0.5], 'spam': [0.1, 0.5, 0.5], 'other': [0.9, 0, 0]})\n    >>> submission = pd.DataFrame({'id': range(3), 'ham': [0.2, 0.3, 0.5], 'spam': [0.1, 0.5, 0.5], 'other': [0.7, 0.2, 0]})\n    >>> score(solution, submission, 'id')\n    0.160531...\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    sample_weights = None\n    if sample_weights_column_name:\n        if sample_weights_column_name not in solution.columns:\n            raise ParticipantVisibleError(f'{sample_weights_column_name} not found in solution columns')\n        sample_weights = solution.pop(sample_weights_column_name)\n\n    if sample_weights_column_name and not micro_average:\n        raise ParticipantVisibleError('Sample weights are only valid if `micro_average` is `True`')\n\n    for col in solution.columns:\n        if col not in submission.columns:\n            raise ParticipantVisibleError(f'Missing submission column {col}')\n\n    kaggle_metric_utilities.verify_valid_probabilities(solution, 'solution')\n    kaggle_metric_utilities.verify_valid_probabilities(submission, 'submission')\n\n    return kaggle_metric_utilities.safe_call_score(kl_divergence, solution, submission, epsilon=epsilon, micro_average=micro_average, sample_weights=sample_weights)\n```\n\n<br>\n\n---\n\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">MY BASIC KL-DIVERGENCE IMPLEMENTATION</b>\n\n```python\nimport numpy as np\nimport pandas as pd\n\ndef kl_divergence(P, Q, epsilon=1e-15):\n    \"\"\"\n    Calculate the Kullback-Leibler Divergence between two probability distributions.\n\n    The function computes the KL Divergence for each column in the dataframes P and Q,\n    and returns the average divergence. It handles the edge cases where probabilities\n    in P are zero by omitting those terms from the summation as per the KL Divergence\n    definition.\n\n    Args:\n        P (pd.DataFrame): A dataframe representing the true probability distribution.\n                          Each column is a separate distribution.\n        Q (pd.DataFrame): A dataframe representing the approximated probability distribution.\n                          Each column should correspond to those in P.\n        epsilon (float): A small value to avoid division by zero and logarithm of zero.\n\n    Returns:\n        float: The average KL Divergence between all distributions in P and Q.\n\n    Raises:\n        ValueError: If the shapes of P and Q don't match.\n\n    \"\"\"\n    if P.shape != Q.shape:\n        raise ValueError(\"The shape of P and Q must be the same.\")\n\n    # Initialize KL Divergence sum\n    kl_sum = 0\n\n    # Iterate over each column (distribution) in the dataframes\n    for col in P.columns:\n        # Extract the probability distributions for the current column\n        p_col = P[col]\n        q_col = Q[col]\n\n        # Clip the probabilities to avoid log(0) and division by zero\n        p_col = np.clip(p_col, epsilon, 1 - epsilon)\n        q_col = np.clip(q_col, epsilon, 1 - epsilon)\n\n        # Compute the KL Divergence for the current column\n        # The divergence is sum(p * log(p/q)) for all elements where p > 0\n        kl_col = np.sum(np.where(p_col != 0, p_col * np.log(p_col / q_col), 0))\n\n        # Add the current column's divergence to the total sum\n        kl_sum += kl_col\n\n    # Compute the average divergence across all columns\n    kl_average = kl_sum / len(P.columns)\n\n    return kl_average\n\n# Example usage\nif __name__ == \"__main__\":\n    # Creating example data\n    P_example = pd.DataFrame({\n        'dist1': [0.25, 0.25, 0.25, 0.25],\n        'dist2': [0.4, 0.3, 0.2, 0.1]\n    })\n\n    Q_example = pd.DataFrame({\n        'dist1': [0.1, 0.2, 0.3, 0.4],\n        'dist2': [0.3, 0.3, 0.2, 0.2]\n    })\n\n    # Calculating KL Divergence\n    kl_result = kl_divergence(P_example, Q_example)\n    print(f\"Average KL Divergence: {kl_result}\")\n\n```\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">2.4 <b>COMPETITION IMPACT INFORMATION</b></h3>\n<hr>\n\n<br>\n\n**Your contribution could significantly improve the accuracy of electroencephalography pattern classification, benefiting neurocritical care, epilepsy treatment, and drug development.**\n\nEEG monitoring currently relies on manual analysis by specialized neurologists, a time-consuming and labor-intensive process prone to errors and variability. **By automating EEG analysis, this competition aims to aid doctors and researchers in detecting seizures and other brain activities more quickly and accurately, potentially leading to faster and more effective treatments.**\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"imports\">3&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"code","source":"print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n!pip install -q pymupdf\n!pip install -q /kaggle/input/kerasv3-lib-ds/tensorflow-2.15.0.post1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install -q /kaggle/input/kerasv3-lib-ds/keras-3.0.2-py3-none-any.whl --no-deps\nimport os; os.environ[\"KERAS_BACKEND\"] = \"jax\"\nprint(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Keras/TF Imports\n\nimport tensorflow as tf; print(f\"\\t\\t- TENSORFLOW VERSION: {tf.__version__}\")\nimport keras; print(f\"\\t\\t- KERAS VERSION: {keras.__version__}\")\nimport keras_cv; print(f\"\\t\\t- KERAS CV VERSION: {keras_cv.__version__}\")\nfrom keras import ops\n\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW-IO VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None);\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\n\n# Built-In Imports (mostly don't worry about these)\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport Levenshtein\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport gc\nimport re\n\n# Visualization Imports (overkill)\nfrom matplotlib.animation import FuncAnimation\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nfrom IPython.display import HTML\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport fitz\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_it_all()\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:25:18.615821Z","iopub.execute_input":"2024-01-28T18:25:18.616066Z","iopub.status.idle":"2024-01-28T18:26:29.818627Z","shell.execute_reply.started":"2024-01-28T18:25:18.616045Z","shell.execute_reply":"2024-01-28T18:26:29.817682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"setup\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"setup\">4&nbsp;&nbsp;SETUP & HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">4.0 FUNCTIONS FROM <b>OTHER KAGGLERS</b> 🩵</h3>\n<hr><br>\n\nTBD\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">4.1 <b>HELPER</b> FUNCTIONS</h3>\n<hr><br>\n\nThese are some functions I carry around with me that I find commonly helpful.","metadata":{}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists into a single list.\n\n    Args:\n        nested_list (Iterable): \n            – A list of lists (or iterables) to be flattened.\n\n    Returns:\n        A flattened list containing all items from the input list of lists.\n    \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\n\ndef print_ln(symbol=\"-\", line_len=110, newline_before=False, newline_after=False):\n    \"\"\" Print a horizontal line of a specified length and symbol.\n\n    Args:\n        symbol (str, optional): \n            – The symbol to use for the horizontal line\n        line_len (int, optional): \n            – The length of the horizontal line in characters\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n            \n    Returns:\n        None; A divider with pre/post new-lines (optional) is printed\n    \"\"\"\n    if newline_before: print();\n    print(symbol * line_len)\n    if newline_after: print();\n        \ndef display_hr(newline_before=False, newline_after=False):\n    \"\"\" Renders a HTML <hr>\n\n    Args:\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n            \n    Returns:\n        None; A divider with pre/post new-lines (optional) is printed\n    \"\"\"\n    if newline_before: print();\n    display(HTML(\"<hr>\"))\n    if newline_after: print();\n        \n\ndef show_pdf(path, dpi=100):\n    \"\"\"\n    Display a PDF file as images within a Jupyter notebook.\n\n    This function opens a PDF file, converts each page to an image, and then\n    displays these images within the notebook.\n\n    Args:\n        path (str): The file path to the PDF document.\n        dpi (int, optional): The desired dots per inch.\n\n    Returns:\n        None: This function does not return anything. It displays images inline.\n    \"\"\"\n    # Open the provided PDF file\n    doc = fitz.open(path)\n\n    # Iterate through each page in the PDF\n    for page in doc:\n        \n        # Render page to a pixmap (an image) at 300 dpi\n        pix = page.get_pixmap(dpi=dpi)\n\n        # Convert the pixmap to a PNG image byte stream\n        img = pix.tobytes(\"png\")\n        \n        # Display the image in the notebook\n        display(Image(img))\n\n    # Close the PDF document to free resources\n    doc.close()\n    \n\ndef save_pdf_as_img(path, save_dir=None, dpi=300):\n    \"\"\"\n    Saves a PDf as an Image\n    \n    Args:\n        path (str): The file path to the PDF document.\n        dpi (int, optional): The desired dots per inch.\n\n    Returns:\n        None: This function does not return anything. It displays images inline.\n    \"\"\"\n    \n    if save_dir is None:\n        save_dir = WORKING_DIR\n    \n    # Open the provided PDF file\n    doc = fitz.open(path)\n\n    # Iterate through each page in the PDF\n    for page in doc:\n        \n        # Render page to a pixmap (an image) at 300 dpi\n        pix = page.get_pixmap(dpi=dpi)\n\n        # Display the image in the notebook\n        pix.save(os.path.join(save_dir, \"img_\"+os.path.basename(path).replace(\"pdf\", \"png\")))\n\n    # Close the PDF document to free resources\n    doc.close()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:26:29.820436Z","iopub.execute_input":"2024-01-28T18:26:29.820699Z","iopub.status.idle":"2024-01-28T18:26:29.832237Z","shell.execute_reply.started":"2024-01-28T18:26:29.820676Z","shell.execute_reply":"2024-01-28T18:26:29.831090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">4.2 <b>LOAD</b> THE DATA</h3>\n<hr><br>\n\nWe also define path information and other constants that are helpful in establishing early.\n\n<br>","metadata":{}},{"cell_type":"code","source":"# ROOT PATHS\nWORKING_DIR = \"/kaggle/working\"\nINPUT_DIR = \"/kaggle/input\"\nCOMPETITION_DIR = os.path.join(INPUT_DIR, \"hms-harmful-brain-activity-classification\")\n\n# ALL COMPETITION DIRECTORIES\nEXAMPLE_FIGURE_DIR = os.path.join(COMPETITION_DIR, \"example_figures\")\nEXAMPLE_IMAGES_DIR = os.path.join(INPUT_DIR, \"hms-example-images\", \"example_figure_images\")\nTRAIN_EEG_DIR = os.path.join(COMPETITION_DIR, \"train_eegs\")\nTRAIN_SPECTROGRAM_DIR = os.path.join(COMPETITION_DIR, \"train_spectrograms\")\nTEST_EEG_DIR = os.path.join(COMPETITION_DIR, \"test_eegs\")\nTEST_SPECTROGRAM_DIR = os.path.join(COMPETITION_DIR, \"test_spectrograms\")\n\n# CORE FILE PATHS\nSS_CSV_PATH = os.path.join(COMPETITION_DIR, \"sample_submission.csv\")\nTRAIN_CSV_PATH = os.path.join(COMPETITION_DIR, \"train.csv\")\nTEST_CSV_PATH = os.path.join(COMPETITION_DIR, \"test.csv\")\n\n# FIGURE STUFF\nEXAMPLE_FIGURE_PDF_PATHS = sorted([os.path.join(EXAMPLE_FIGURE_DIR, fname) for fname in os.listdir(EXAMPLE_FIGURE_DIR)], key=lambda x: int(x[-6:-4]))\nEXAMPLE_IMAGE_PATHS = sorted([os.path.join(EXAMPLE_IMAGES_DIR, fname) for fname in os.listdir(EXAMPLE_IMAGES_DIR)], key=lambda x: int(x[-6:-4]))\n\n# CORE OBJECTS\ndisplay_hr(newline_before=True, newline_after=True)\nprint(\"\\n... SAMPLE SUBMISSION DATAFRAME ...\\n\\n\")\nss_df = pd.read_csv(SS_CSV_PATH)\ndisplay(ss_df)\ndisplay_hr(newline_before=True, newline_after=True)\nprint(\"\\n... TRAIN DATAFRAME ...\\n\\n\")\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ntrain_df[\"eeg_path\"] = train_df[\"eeg_id\"].apply(lambda x: os.path.join(TRAIN_EEG_DIR, f\"{x}.parquet\"))\ntrain_df[\"spectrogram_path\"] = train_df[\"spectrogram_id\"].apply(lambda x: os.path.join(TRAIN_SPECTROGRAM_DIR, f\"{x}.parquet\"))\ndisplay(train_df.info())\ndisplay(train_df.describe())\ndisplay(train_df)\ndisplay_hr(newline_before=True, newline_after=True)\nprint(\"\\n... TEST DATAFRAME ...\\n\\n\")\ntest_df = pd.read_csv(TEST_CSV_PATH)\ntest_df[\"eeg_path\"] = test_df[\"eeg_id\"].apply(lambda x: os.path.join(TEST_EEG_DIR, f\"{x}.parquet\"))\ntest_df[\"spectrogram_path\"] = test_df[\"spectrogram_id\"].apply(lambda x: os.path.join(TEST_SPECTROGRAM_DIR, f\"{x}.parquet\"))\ndisplay(test_df.info())\ndisplay(test_df.describe())\ndisplay(test_df)\ndisplay_hr(newline_before=True, newline_after=True)\n\n# Helpers for later anbd plotting\nLIGHT_COLOR = '#77AAB0'\nDARK_COLOR = '#1B5264'","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:27:21.226866Z","iopub.execute_input":"2024-01-28T18:27:21.227268Z","iopub.status.idle":"2024-01-28T18:27:22.137048Z","shell.execute_reply.started":"2024-01-28T18:27:21.227237Z","shell.execute_reply":"2024-01-28T18:27:22.136192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"eda\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"eda\">5&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\nIt's pretty clear that we have grouping of information which we would like to explore.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">GROUP 1: EEG DATA</b>\n\n* **`eeg_id`**\n* **`eeg_sub_id`**\n* **`eeg_label_offset_seconds`**\n* **`eeg_path`**\n\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">GROUP 2: SPECTROGRAM DATA</b>\n\n* **`spectrogram_id`**\n* **`spectrogram_sub_id`**\n* **`spectrogram_label_offset_seconds`**\n* **`spectrogram_path`**\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">GROUP 3: LABEL DATA</b>\n\n* **`label_id`**\n* **`expert_consensus`**\n* **`<CLS>_vote`**\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\">GROUP 4: OTHER COLUMNS</b>\n\n* **`patient_id`**","metadata":{}},{"cell_type":"code","source":"train_df.expert_consensus.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:52:26.901516Z","iopub.execute_input":"2024-01-28T18:52:26.902400Z","iopub.status.idle":"2024-01-28T18:52:26.926130Z","shell.execute_reply.started":"2024-01-28T18:52:26.902357Z","shell.execute_reply":"2024-01-28T18:52:26.925046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">5.1 <b>GROUP 1:</b> EEG DATA</h3>\n<hr><br>\n\n**COLUMN SPECIFIC INFORMATION**\n\n**`eeg_id`**\n- HOST PROVIDED INFO:\n    - **`eeg_id`** is a unique identifier for the entire EEG recording. \n    - This ID represents a specific EEG dataset, encompassing all the data collected during one recording session for a patient.\n- ADDITIONAL INFO:\n    - **Purpose**: In a dataset, `eeg_id` helps in segregating data from different recording sessions, ensuring that analyses or machine learning models correctly associate all data fragments from the same session.\n    - **Data Handling**: When processing EEG data, it's crucial to maintain the association between `eeg_id` and the corresponding EEG readings to correctly interpret the results.\n\n**`eeg_sub_id`**\n- HOST PROVIDED INFO:\n    - **`eeg_sub_id`** is an identifier for a specific 50-second long subsample of the EEG recording. This ID is crucial for referencing the exact segment of EEG data that the annotations or labels apply to.\n- ADDITIONAL INFO:\n    - **Segmentation Relevance**: EEG data is often segmented for detailed analysis. `eeg_sub_id` facilitates the study of specific time windows, which is essential in seizure detection and classification.\n    - **Overlap and Consolidation**: Given that many of these samples overlapped and have been consolidated, `eeg_sub_id` assists in navigating these overlaps for accurate data interpretation.\n\n**`eeg_label_offset_seconds`**\n- HOST PROVIDED INFO:\n    - **`eeg_label_offset_seconds`** indicates the time between the beginning of the consolidated EEG and the specific subsample. This is crucial for locating the exact portion of the EEG recording that the labels refer to.\n- ADDITIONAL INFO:\n    - **Timing Accuracy**: Accurate timing is critical in EEG analysis, especially for seizure detection. This offset ensures precise temporal alignment between the labels and the EEG data.\n    - **Synchronization**: In studies involving multimodal data (like EEG and EKG), this offset helps synchronize EEG data with other types of recordings.\n\n**`eeg_path`**\n- HOST PROVIDED INFO:\n    - `eeg_path` was added by us during loading to provide a way of directly accessing the relevant information.\n- ADDITIONAL INFO:\n    - **Data Access**: `eeg_path` is used to access the EEG data files for processing and analysis. It's crucial for loading the data into analysis software or scripts.\n\n---\n\n<br>\n\n**THINGS WE NEED TO EXPLORE...**\n\n- **Data Quality**: \n    - We should check that the EEG data is free from artifacts and noise.\n    - If the dataset is NOT free from this, we should understand these aberrations to better understand the distribution they come from\n- **Feature Extraction**: \n    - Techniques like Fourier Transform or Wavelet Transform are often used to extract features from EEG data for seizure detection. \n    - Can we use similar techniques?\n- **Machine Learning Approaches**: \n    - What machine learning architectures are commonly used for tasks like this?\n    - What are the open-source contributions in this competition that we can learn from?\n    \n<br>","metadata":{}},{"cell_type":"code","source":"# PLOT to show the distribution of datapoints per eeg recording\nplt.figure(figsize=(18, 10))\nBIN_SIZE = 100\neeg_recording_counts = train_df['eeg_id'].value_counts()\nplt.hist(eeg_recording_counts, bins=BIN_SIZE, log=True, color=LIGHT_COLOR)\nplt.xlabel('Number of Rows per EEG ID', fontweight=\"bold\")\nplt.ylabel('Number of EEG IDs (LOG SCALE)', fontweight=\"bold\")\nplt.title('Distribution of Number of Rows per EEG ID', fontweight=\"bold\")\nplt.show()\n\n# PLOT to show the top 50 recordings with the most data points\nTOP_N = 50\ntop_ids = eeg_recording_counts.head(TOP_N)\nplt.figure(figsize=(18, 10))\ntop_ids.plot(kind='bar', color=DARK_COLOR)\nplt.xlabel('EEG ID', fontweight=\"bold\")\nplt.ylabel('Number of Rows', fontweight=\"bold\")\nplt.title(f'Top {TOP_N} EEG IDs by Number of Rows', fontweight=\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:41:21.213769Z","iopub.execute_input":"2024-01-28T18:41:21.214347Z","iopub.status.idle":"2024-01-28T18:41:22.676503Z","shell.execute_reply.started":"2024-01-28T18:41:21.214298Z","shell.execute_reply":"2024-01-28T18:41:22.675630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import welch\nfrom scipy.signal import spectrogram\n\ndemo_df = pd.read_parquet(train_df[train_df.eeg_id==1628180742].eeg_path[0])\n# Select the electrodes columns, assuming they are labeled as in your example\n\nelectrodes = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2']\n\n# Determine the number of subplots needed\nn_electrodes = len(electrodes)\nn_rows = n_electrodes\n\ndef interpolate_color(color1, color2, num_colors):\n    \"\"\"Interpolate between two hex color values, returning a list of num_colors colors.\"\"\"\n    color1_rgb = matplotlib.colors.hex2color(color1)\n    color2_rgb = matplotlib.colors.hex2color(color2)\n    return [matplotlib.colors.rgb2hex([s + (e - s) * i / (num_colors - 1) for s, e in zip(color1_rgb, color2_rgb)]) for i in range(num_colors)]\n\n# Define your color story\nstart_color = '#77AAB0'  # Lighter shade\nend_color = '#1B5264'    # Darker shade\n\n# Create a list of colors\ncolors = interpolate_color(start_color, end_color, n_electrodes)\n\n# Create subplots\nfig, axes = plt.subplots(n_rows, 1, figsize=(15, 2 * n_rows), sharex=True)\n\n# Now plotting with the color map\nfor i, electrode in enumerate(electrodes):\n    axes[i].plot(demo_df.index, demo_df[electrode], label=f\"{electrode}\", color=colors[i])\n    axes[i].set_title(f'EEG Signal - {electrode}', fontweight='bold')\n    axes[i].set_ylabel('Amp', fontweight='bold')\n\nplt.xlabel('Time (samples)', fontweight='bold')\nplt.show()\n\nf, Pxx = welch(demo_df['Fp1'], fs=1)  # fs is the sampling frequency\nplt.semilogy(f, Pxx)\nplt.xlabel('Frequency [Hz]')\nplt.ylabel('PSD [V**2/Hz]')\nplt.title('Power Spectral Density for Channel Fp1')\nplt.show()\n\n# Example for channel Fp1\nf, t, Sxx = spectrogram(demo_df['Fp1'], fs=1)  # fs is the sampling frequency\nplt.pcolormesh(t, f, 10 * np.log10(Sxx))\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.title('Spectrogram for Channel Fp1')\nplt.colorbar(label='Intensity [dB]')\nplt.show()\n\n\nplt.figure(figsize=(18, 10))\nsns.heatmap(demo_df.T, cmap='viridis')\nplt.xlabel('Time', fontweight='bold')\nplt.ylabel('Channels', fontweight='bold')\nplt.title('Heatmap of EEG Channel Amplitudes Over Time', fontweight='bold')\nplt.show()\n\nplt.figure(figsize=(18, 10))\nplt.plot(demo_df['Fp1'], label='Fp1')\nplt.plot(demo_df['F3'], label='F3')  # Add more channels as needed\nplt.xlabel('Time', fontweight='bold')\nplt.ylabel('Amplitude', fontweight='bold')\nplt.title('Comparison of Different EEG Channels', fontweight='bold')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:06:16.814545Z","iopub.execute_input":"2024-01-28T19:06:16.814905Z","iopub.status.idle":"2024-01-28T19:06:24.478476Z","shell.execute_reply.started":"2024-01-28T19:06:16.814875Z","shell.execute_reply":"2024-01-28T19:06:24.477540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.interpolate import griddata\nIMG_URL = \"https://upload.wikimedia.org/wikipedia/commons/thumb/7/70/21_electrodes_of_International_10-20_system_for_EEG.svg/1200px-21_electrodes_of_International_10-20_system_for_EEG.svg.png\"\nIMG_PATH = \"/kaggle/working/1200px-21_electrodes_of_International_10-20_system_for_EEG.svg.png\"\nif not os.path.isfile(IMG_PATH):\n    !wget {IMG_URL}\n    \ndef load_background(image_path):\n    \"\"\"\n    Load an image from a given path and return it.\n\n    Args:\n        image_path (str): The path to the image file.\n\n    Returns:\n        Image: The loaded image.\n    \"\"\"\n    # Load and return the background image\n    return Image.open(image_path)\n\ndef create_interpolated_grid(image, electrode_positions, values, grid_resolution=200):\n    \"\"\"\n    Interpolate electrode values over a specified grid resolution and return the interpolated grid.\n\n    Args:\n        image (Image): The background image over which to overlay the grid.\n        electrode_positions (dict): A dictionary containing electrode names as keys and positions as values.\n        values (np.ndarray): An array of values associated with each electrode.\n        grid_resolution (int, optional): The resolution of the interpolation grid. Defaults to 200.\n\n    Returns:\n        np.ndarray: The interpolated grid of values over the image area.\n    \"\"\"\n    # Define the grid for interpolation based on the image size and desired resolution\n    grid_x, grid_y = np.mgrid[0:image.size[0]:complex(grid_resolution), 0:image.size[1]:complex(grid_resolution)]\n\n    # Convert electrode positions from dictionary to an array of (x, y) pairs\n    positions = np.array([pos for pos in electrode_positions.values()])\n\n    # Interpolate the electrode values onto the grid\n    return griddata(positions, values, (grid_x, grid_y), method='cubic')\n\ndef plot_heatmap_on_image(image, grid_z):\n    \"\"\"\n    Plot an interpolated heatmap on top of a background image.\n\n    Args:\n        image (Image): The background image over which to overlay the heatmap.\n        grid_z (np.ndarray): The interpolated grid of values to use as the heatmap.\n    \"\"\"\n    # Create a figure and axis to plot the background image\n    fig, ax = plt.subplots(figsize=(18, 9))\n    ax.imshow(image, extent=[0, image.size[0], 0, image.size[1]])\n\n    # Overlay the heatmap on top of the background image\n    # The grid_z data must be transposed due to the different coordinate systems of the image and the grid\n    im = ax.imshow(grid_z.T, extent=(0, image.size[0], 0, image.size[1]), \n                   origin='lower', cmap='viridis', alpha=0.6)\n\n    # Add a color bar for the interpolated heatmap\n    cbar = plt.colorbar(im, ax=ax, orientation='vertical', fraction=0.02, pad=0.04)\n    cbar.set_label('Heatmap Values', rotation=270, labelpad=15)\n\n    # Remove axis for better visualization\n    ax.axis('off')\n\n    # Display the plot\n    plt.show()\n\n    \ndef update_electrode_positions(image, electrode_positions):\n    \"\"\"\n    Update electrode positions by converting percentages to absolute pixel values based on the image size.\n\n    Args:\n        image (Image): The background image used to determine the size for position scaling.\n        electrode_positions (dict): A dictionary containing electrode names as keys and positions as percentage values.\n\n    Returns:\n        dict: A dictionary with electrode positions updated to absolute pixel values.\n    \"\"\"\n    # Get the size of the image\n    img_width, img_height = image.size\n\n    # Calculate absolute positions from percentages\n    if list(electrode_positions.values())[0][0]>1:\n        absolute_positions = {\n            elec: (int(pos[0]/100 * img_width), int(pos[1]/100 * img_height))\n            for elec, pos in electrode_positions.items()\n        }\n    else:\n        absolute_positions = {\n            elec: (int(pos[0] * img_width), int(pos[1] * img_height))\n            for elec, pos in electrode_positions.items()\n        }\n\n    return absolute_positions\n\n\ndef plot_electrode_overlay(image_path, electrode_percentage_positions):\n    \"\"\"\n    Plot the EEG image and overlay it with dots at the electrode positions.\n\n    Args:\n        image_path (str): The path to the EEG image file.\n        electrode_percentage_positions (dict): A dictionary containing electrode names as keys and positions as percentage values.\n    \"\"\"\n    # Load the image\n    image = Image.open(image_path)\n    img_width, img_height = image.size\n\n    # Create a plot\n    fig, ax = plt.subplots()\n    ax.imshow(image)\n\n    # Overlay the dots at the percentage-based electrode positions\n    for label, (percent_x, percent_y) in electrode_percentage_positions.items():\n        # Convert percentage to absolute coordinates\n        x_pixel = img_width * percent_x / 100\n        y_pixel = img_height * percent_y / 100\n\n        # Plot the electrode position\n        ax.plot(x_pixel, y_pixel, 'ro')  # Red dot\n        ax.text(x_pixel, y_pixel, label, fontsize=8, ha='center', va='center', color='white')\n\n    # Hide the axes\n    ax.axis('off')\n\n    # Show the plot\n    plt.show()\n    \n    \n    \n# 10/20 Electrode Image\nbackground_img = load_background(IMG_PATH)\n\n# Electrode positions on a 2D plane, these positions will be estimates based on the provided image\nelectrode_positions = {\n 'Fp1': (40.0, 83.5),\n 'Fp2': (60.0, 83.5),\n 'F7': (25.0, 70),\n 'F3': (99.0, 99.26),\n 'Fz': (99.0, 99.26),\n 'F4': (99.0, 99.26),\n 'F8': (99.0, 99.26),\n 'T3': (99.0, 99.26),\n 'C3': (99.0, 99.26),\n 'Cz': (99.0, 99.26),\n 'C4': (99.0, 99.26),\n 'T4': (99.0, 99.26),\n 'T5': (99.0, 99.26),\n 'P3': (99.0, 99.26),\n 'Pz': (99.0, 99.26),\n 'P4': (38.33, 45.85),\n 'T6': (42.67, 45.85),\n 'O1': (21.67, 57.32),\n 'O2': (38.33, 57.32),\n #'A1': (8.67, 34.39),\n #'A2': (51.33, 34.39)\n}\n\nplot_electrode_overlay(IMG_PATH, electrode_positions)\n# # Example: Calculating mean intensity for each electrode\n# mean_intensities = {electrode: demo_df[electrode].mean() for electrode in electrodes}\n# values = np.array([mean_intensities[elec] for elec in electrodes])\n\n# # Create the interpolated grid based on the electrode positions and values\n# grid_z = create_interpolated_grid(background_img, update_electrode_positions(background_img, electrode_positions), values)\n\n# print(update_electrode_positions(background_img, electrode_positions))\n\n# # Plot the heatmap on top of the image\n# plot_heatmap_on_image(background_img, grid_z)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:36:07.037710Z","iopub.execute_input":"2024-01-28T19:36:07.038168Z","iopub.status.idle":"2024-01-28T19:36:07.414790Z","shell.execute_reply.started":"2024-01-28T19:36:07.038120Z","shell.execute_reply":"2024-01-28T19:36:07.413978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"electrode_positions = {\n    'Fp1': (-1, 4), 'Fp2': (1, 4),\n    'F3': (-2, 3), 'F4': (2, 3),\n    'C3': (-2, 1), 'C4': (2, 1),\n    'P3': (-2, -1), 'P4': (2, -1),\n    'O1': (-1, -3), 'O2': (1, -3),\n    'F7': (-3, 3), 'F8': (3, 3),\n    'T3': (-3, 1), 'T4': (3, 1),\n    'T5': (-3, -1), 'T6': (3, -1),\n    'Fz': (0, 3), 'Cz': (0, 1), 'Pz': (0, -1)\n}\n\n# Example: Calculating mean intensity for each electrode\nmean_intensities = {electrode: demo_df[electrode].mean() for electrode in electrodes}\n\nfrom scipy.interpolate import griddata\n\n# Create grid coordinates\ngrid_x, grid_y = np.mgrid[-4:4:100j, -4:4:100j]\n\n# Create an array of electrode positions and their corresponding mean values\npositions = np.array([electrode_positions[elec] for elec in electrodes])\nvalues = np.array([mean_intensities[elec] for elec in electrodes])\n\n# Interpolate\ngrid_z = griddata(positions, values, (grid_x, grid_y), method='cubic')\n\n# Plot\nplt.imshow(grid_z.T, extent=(-4,4,-4,4), origin='lower')\nplt.colorbar(label='Intensity')\nplt.title('EEG Electrode Activity Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T19:15:58.342969Z","iopub.execute_input":"2024-01-28T19:15:58.343292Z","iopub.status.idle":"2024-01-28T19:15:58.710057Z","shell.execute_reply.started":"2024-01-28T19:15:58.343266Z","shell.execute_reply":"2024-01-28T19:15:58.709186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT to show the distribution of subsamples per eeg recording\nplt.figure(figsize=(18, 10))\nsub_id_counts = train_df.groupby('eeg_id')['eeg_sub_id'].nunique()\nplt.hist(sub_id_counts, bins=50, log=True, color='#77AAB0')\nplt.xlabel('Number of Unique Subsamples per EEG ID', fontweight='bold')\nplt.ylabel('Number of EEG IDs', fontweight='bold')\nplt.title('Distribution of Subsamples per EEG Recording', fontweight='bold')\nplt.show()\n\nDEMO_EEG_ID = '1628180742'\nDEMO_DATA = train_df[train_df['eeg_id'] == DEMO_EEG_ID]","metadata":{"execution":{"iopub.status.busy":"2024-01-28T18:42:31.940159Z","iopub.execute_input":"2024-01-28T18:42:31.940977Z","iopub.status.idle":"2024-01-28T18:42:32.693539Z","shell.execute_reply.started":"2024-01-28T18:42:31.940946Z","shell.execute_reply":"2024-01-28T18:42:32.692676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">5.2 <b>GROUP 2:</b> SPECTROGRAM DATA</h3>\n<hr><br>\n\n**COLUMN SPECIFIC INFORMATION**\n\n**`spectrogram_id`**\n- Host Provided Information:\n    - **`spectrogram_id`** is a unique identifier for the entire spectrogram recording associated with EEG data. Spectrograms are visual representations of the spectrum of frequencies in a signal as they vary with time.\n- Additional Relevant Information:\n    - **Spectrogram Utility**: In EEG analysis, spectrograms are useful for identifying changes in brain activity over time, which is particularly relevant for seizure detection.\n    - **Data Association**: Just like `eeg_id`, `spectrogram_id` ensures that all segments of a spectrogram are correctly associated with their respective EEG recording sessions.\n\n**`spectrogram_sub_id`**\n- Host Provided Information:\n    - **`spectrogram_sub_id`** is an identifier for a specific segment of the spectrogram, analogous to `eeg_sub_id` for EEG data. This is particularly important for the 10-minute subsamples mentioned in the dataset.\n- Additional Relevant Information:\n    - **Segmentation and Analysis**: The segmentation of spectrograms into smaller, manageable parts allows for more detailed and focused analysis of specific time periods within the EEG data.\n    - **Overlap Management**: This ID is crucial for handling any overlapping segments within the spectrogram data, ensuring accurate mapping and analysis of these overlaps.\n\n**`spectrogram_label_offset_seconds`**\n- Host Provided Information:\n    - **`spectrogram_label_offset_seconds`** indicates the time offset between the start of the consolidated spectrogram and the subsample in question. This is key for pinpointing the exact portion of the spectrogram that corresponds to the annotated EEG data.\n- Additional Relevant Information:\n    - **Precision in Time Alignment**: Accurate time alignment is critical in synchronizing the spectrogram data with its corresponding EEG signals, especially when analyzing temporal patterns like seizures.\n    - **Multimodal Data Synchronization**: This offset is also important when the spectrogram is used in conjunction with other data types (like EEG or EKG), ensuring all data types are properly aligned temporally.\n\n**`spectrogram_path`**\n- Host Provided Information:\n    - **`spectrogram_path`** was added by us during loading to provide a way of directly accessing the relevant information.\n- Additional Relevant Information:\n    - **Data Retrieval**: This path is essential for accessing the spectrogram files for analysis, which might be stored in various formats like images or binary files.\n\n---\n\n<br>\n\n**THINGS WE NEED TO EXPLORE...**\n\n- **Visualization and Interpretation**: \n    - How can spectrograms provide a more intuitive way to interpret EEG data, especially for frequency changes over time?\n    - Are there specific patterns in spectrograms that are indicative of seizures or other abnormal brain activities?\n- **Frequency Analysis Techniques**: \n    - Which techniques are most effective for analyzing the frequency characteristics in spectrograms related to EEG data?\n    - Can these techniques be integrated with traditional EEG signal analysis for a more comprehensive understanding?\n- **Computational Considerations**: \n    - What are the computational challenges in processing and analyzing large sets of spectrogram data?\n    - Are there specific tools or software that are particularly effective for this kind of analysis?\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">5.3 <b>GROUP 3:</b> LABEL DATA</h3>\n<hr><br>\n\nWe are trying to detect and classify seizures and other types of harmful brain activity in electroencephalography (EEG) data. As such, we are provided with **labelled data** in the form of expert votes (1-28(!) votes depending on the patient/case) along with a consensus column (self-explanatory) and label ID. \n\nBefore we dive into the individual columns, let us better understand the labels themselves as this will help us understand the columns better.\n\nUnderstanding the labels and classifications in EEG data for seizure detection and other brain activities is crucial for accurate diagnosis and treatment planning. Here's a more detailed breakdown of each label, along with insights on how EEGs and spectrograms play a role in this context.\n\n**NOTE: The images below are from supplementary materials (external to the competition) and not the provided PDFs. We will visualize each label from the PDFs in the code cells below.**\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">SEIZURE</span></b>\n\n**CHARACTERISTICS IN EEG**: \n* Seizures typically appear as sudden bursts or spikes of electrical activity that stand out from the normal brain wave patterns. \n* In focal seizures, these abnormal patterns are localized to one area, whereas in generalized seizures, they appear across both hemispheres.\n\n**SPECTROGRAM UNDERSTANDING**: \n* On a spectrogram, seizures may show as intense, localized bursts of power, especially in higher frequency bands. \n* The time of onset, duration, and spread can be visualized.\n\n**EEG IMAGES**:\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_seizure_1.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_seizure_2.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_seizure_3.png?raw=true\"></center>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 15px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">LPD</span> (Lateralized Periodic Discharges)</b>\n\n**CHARACTERISTICS IN EEG**: \n* LPDs are repetitive and localized to one hemisphere. \n* They may present as sharp waves or spikes and can be periodic or quasi-periodic. \n* The regularity, morphology, and interval of discharges are key aspects.\n\n**SPECTROGRAM UNDERSTANDING**: \n* LPDs will typically show up as rhythmic patterns at consistent intervals, predominantly in one hemisphere. \n* The spectrogram can help in determining the frequency of these discharges and their evolution over time.\n\n**EEG IMAGES**:\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lpd_1.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lpd_2.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lpd_3.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lpd_4.png?raw=true\"></center>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">GPD</span> (Generalized Periodic Discharges)</b>\n\n**CHARACTERISTICS IN EEG**: \n* GPDs are similar to LPDs but are generalized across both hemispheres. \n* The pattern can be symmetric or asymmetric and can vary in frequency and amplitude.\n\n**SPECTROGRAM UNDERSTANDING**: \n* In spectrograms, GPDs appear as rhythmic patterns occurring at regular intervals across both hemispheres. \n* Analysis might reveal the uniformity and spread of these discharges.\n\n**EEG IMAGES**:\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_gpd_1.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_gpd_2.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_gpd_3.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_gpd_4.png?raw=true\"></center>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">LRDA</span> (Lateralized Rhythmic Delta Activity)</b>\n\n**CHARACTERISTICS IN EEG**: \n* LRDA is characterized by slower, rhythmic delta waves that are confined to one hemisphere.\n* These waves are often associated with destructive lesions or other focal brain disturbances.\n\n**SPECTROGRAM UNDERSTANDING**: \n* LRDA will manifest as continuous or quasi-continuous waves in the delta frequency range (1-4 Hz), predominantly in one hemisphere. \n* The spectrogram helps in visualizing the consistency and localization of these waves.\n\n**EEG IMAGES**:\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lrda_1.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_lrda_2.png?raw=true\"></center>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">GRDA</span> (Generalized Rhythmic Delta Activity)</b>\n\n**CHARACTERISTICS IN EEG**: \n* GRDA involves delta waves like LRDA, but they are generalized across both cerebral hemispheres. \n* These patterns are usually seen in more severe and diffuse brain disturbances.\n\n**SPECTROGRAM UNDERSTANDING**: \n* GRDA will be evident as rhythmic delta activity across both hemispheres. \n* The spectrogram can be used to assess the extent and uniformity of this activity.\n\n**EEG IMAGES**:\n\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_grda_1.png?raw=true\"></center>\n<center><img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/eeg_grda_2.png?raw=true\"></center>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">Other</span></b>\n\n**CHARACTERISTICS IN EEG**: \n* This category encompasses various other EEG patterns that don't fit into the above classifications. \n* It may include triphasic waves, burst suppression patterns, or other atypical periodic or rhythmic patterns.\n\n**SPECTROGRAM UNDERSTANDING**: \n* This category will require a more nuanced analysis as it includes a diverse range of patterns. \n* The spectrogram can assist in identifying unique or atypical features not covered by standard classifications.\n\n**EEG IMAGES**:\n\nTOO MANY IMAGES TO PUT HERE... BUT I PULLED MY IMAGES FROM <b><a href=\"http://links.lww.com/JCNP/A134\" style=\"color: red;\">HERE</a></b>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; font-size: 18px; text-transform: uppercase; letter-spacing: 2px;\"><span style=\"color: #1B5264;\">NOTE:</span> ABOUT USING EEGS AND SPECTROGRAMS IN ANALYSIS:</b>\n\n**EEG ANALYSIS**: \n* Provides real-time electrical activity of the brain, essential for identifying the type, location, and dynamics of abnormal brain activity.\n\n**SPECTROGRAM ANALYSIS**: \n* Offers a visual representation of the spectrum of frequencies of a signal as it varies with time. \n* This is particularly useful in identifying rhythmic or periodic patterns and their evolution over time.\n\n**INTER-EXPERT VARIABILITY**: \n* We need to be aware that even among experts, there can be variability in interpreting EEG patterns. \n* Our analysis should consider this variability as a factor in decision-making.","metadata":{}},{"cell_type":"markdown","source":"<a id=\"baseline\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"baseline\">6&nbsp;&nbsp;BASELINE&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">6.1 <b>PLACE</b> HOLDER</h3>\n<hr><br>\n<ul>\n        <li>Placeholder 1</li>\n        <li>Placeholder 2</li>\n        <li>Placeholder 3</li>\n    </ul>\n<br>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"next_steps\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"next_steps\">7&nbsp;&nbsp;NEXT STEPS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #1B5264; background-color: #ffffff;\">7.1 <b>PLACE</b> HOLDER</h3>\n<hr><br>\n<ul>\n        <li>Placeholder 1</li>\n        <li>Placeholder 2</li>\n        <li>Placeholder 3</li>\n    </ul>\n<br>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"appendix_label_images\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #77AAB0;\" id=\"appendix_label_images\">8&nbsp;&nbsp;APPENDIX - LABEL IMAGES&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #77AAB0;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\n| **PDF Title**        | **PDF Content**                                                                                                       |\n|:--------------------:|------------------------------------------------------------------------------------------------------------------|\n| **Image Sample 01**| ![Image Sample 01](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample01.png?raw=true)        |\n| **Image Sample 02**| ![Image Sample 02](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample02.png?raw=true)        |\n| **Image Sample 03**| ![Image Sample 03](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample03.png?raw=true)        |\n| **Image Sample 04**| ![Image Sample 04](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample04.png?raw=true)        |\n| **Image Sample 05**| ![Image Sample 05](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample05.png?raw=true)        |\n| **Image Sample 06**| ![Image Sample 06](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample06.png?raw=true)        |\n| **Image Sample 07**| ![Image Sample 07](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample07.png?raw=true)        |\n| **Image Sample 08**| ![Image Sample 08](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample08.png?raw=true)        |\n| **Image Sample 09**| ![Image Sample 09](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample09.png?raw=true)        |\n| **Image Sample 10**| ![Image Sample 10](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample10.png?raw=true)        |\n| **Image Sample 11**| ![Image Sample 11](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample11.png?raw=true)        |\n| **Image Sample 12**| ![Image Sample 12](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample12.png?raw=true)        |\n| **Image Sample 13**| ![Image Sample 13](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample13.png?raw=true)        |\n| **Image Sample 14**| ![Image Sample 14](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample14.png?raw=true)        |\n| **Image Sample 15**| ![Image Sample 15](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample15.png?raw=true)        |\n| **Image Sample 16**| ![Image Sample 16](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample16.png?raw=true)        |\n| **Image Sample 17**| ![Image Sample 17](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample17.png?raw=true)        |\n| **Image Sample 18**| ![Image Sample 18](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample18.png?raw=true)        |\n| **Image Sample 19**| ![Image Sample 19](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample19.png?raw=true)        |\n| **Image Sample 20**| ![Image Sample 20](https://github.com/darien-schettler/asset-hosting/blob/main/img_Sample20.png?raw=true)        |\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}