{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"It is always important to look at the data and figure out what kind of data we are going to be working with.","metadata":{}},{"cell_type":"markdown","source":"# Prep","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:21.043535Z","iopub.execute_input":"2024-01-14T10:37:21.043958Z","iopub.status.idle":"2024-01-14T10:37:21.050810Z","shell.execute_reply.started":"2024-01-14T10:37:21.043922Z","shell.execute_reply":"2024-01-14T10:37:21.049407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show EEG and Spctr parallelly\ndef show_eeg_spctr(df_eeg, df_spctr, title=''):\n    plt.rcParams[\"font.size\"] = 32\n    #set time axis\n    t_eeg = np.arange(0, df_eeg.shape[0]*0.01, 0.01) # step = 0.01s/slice, df_eeg.shape[0]: 5000 entries for 50 s EEG records, \n    t_spctr = np.arange(0, df_spctr.shape[0]/30, 1/30) # step = 1/30min (2s) /slice, df_spctr.shape[0]: 300 entries for 10 min == 600 s spectrograms records\n    \n    # Make fig and ax\n    fig, (ax1, ax2) = plt.subplots(nrows=2,ncols=1, figsize=(30, 30),tight_layout=True)\n    \n    # EEG plotting\n    for i,col in enumerate(df_eeg): #plot each columns\n        y = df_eeg[col]\n        ax1.plot(t_eeg, y, color=cm.hsv(i/df_eeg.shape[1]), label=col, alpha=0.3) # color - for gradient coloring\n    ax1.set(title=f'{title}_EEG',xlabel='Time (s)', ylabel='intensity',ylim=(-250,250))\n    ax1.legend(loc='center left', bbox_to_anchor=(1, .5), ncols=1) # legend loc adjustment\n    \n    # spectrogram plotting\n    for i,col in enumerate(df_spctr.iloc[:, 1:]): # first column: time\n        y = df_spctr[col]\n        if i%25 == 0: #Filter labels name on legend\n            ax2.plot(t_spctr, y, color=cm.hsv(i/300.0), label=col, alpha=0.3) # color - Circular gradient so that colors correspond to angular locations\n        else:\n            ax2.plot(t_spctr, y, color=cm.hsv(i/300.0), alpha=0.15)\n    ax2.set(title =f'{title}_Spectrogram', xlabel='Time (min)', ylabel='intensity', ylim=(0,500))\n    ax2.legend(loc='center left', bbox_to_anchor=(1, .5), ncols=1) # legend loc adjustment\n    \n    plt.show()\n    plt.rcParams[\"font.size\"] = 12","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-01-14T10:37:59.480141Z","iopub.execute_input":"2024-01-14T10:37:59.480553Z","iopub.status.idle":"2024-01-14T10:37:59.496135Z","shell.execute_reply.started":"2024-01-14T10:37:59.480520Z","shell.execute_reply":"2024-01-14T10:37:59.493686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test_csv\n> Only **spectrogram_id, eeg_id** and **patient_id** are given - because there will be no duplication in EEG and spectrograms.","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\ntest","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:23.013504Z","iopub.execute_input":"2024-01-14T10:37:23.013924Z","iopub.status.idle":"2024-01-14T10:37:23.029490Z","shell.execute_reply.started":"2024-01-14T10:37:23.013891Z","shell.execute_reply":"2024-01-14T10:37:23.028044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train_csv\n> \"Metadata for the train set. The expert annotators reviewed 50 second long EEG samples plus matched spectrograms covering 10 a minute window centered at the same time and labeled the central 10 seconds. Many of these samples overlapped and have been consolidated. train.csv provides the metadata that allows you to extract the original subsets that the raters annotated.\"","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:24.128342Z","iopub.execute_input":"2024-01-14T10:37:24.128751Z","iopub.status.idle":"2024-01-14T10:37:24.378231Z","shell.execute_reply.started":"2024-01-14T10:37:24.128717Z","shell.execute_reply":"2024-01-14T10:37:24.377143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**unique values counter**","metadata":{}},{"cell_type":"code","source":"unique_values = pd.DataFrame()\nfor col in train.columns[:-6]:\n    unique_values.at['unique_count', col] = len(train[col].unique())\nunique_values.astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:24.647405Z","iopub.execute_input":"2024-01-14T10:37:24.648836Z","iopub.status.idle":"2024-01-14T10:37:24.699228Z","shell.execute_reply.started":"2024-01-14T10:37:24.648769Z","shell.execute_reply":"2024-01-14T10:37:24.697929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### eeg_id \n> **\"A unique identifier for the entire EEG recording.\"**  - Correspond to train_eegs","metadata":{}},{"cell_type":"code","source":"# load eeg by id\neeg0_id = train['eeg_id'][0]\neeg0 = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{eeg0_id}.parquet')\neeg0","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:25.044245Z","iopub.execute_input":"2024-01-14T10:37:25.044669Z","iopub.status.idle":"2024-01-14T10:37:25.090714Z","shell.execute_reply.started":"2024-01-14T10:37:25.044635Z","shell.execute_reply":"2024-01-14T10:37:25.089546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### eeg_sub_id\n> **\"An ID for the specific 50 second long subsample this row's labels apply to.\"** - some train_eegs are recorded longer and More subsamples are extracted","metadata":{}},{"cell_type":"code","source":"#number of subsamples per sample\neeg_sub_id_num = pd.DataFrame(\n    np.zeros((len(train['eeg_sub_id'].unique()), 1)), columns=['number_of_ids']\n)\nfor i in range(len(train['eeg_sub_id'].unique())):\n    num = train[train['eeg_sub_id']==i].shape[0]\n    eeg_sub_id_num.iloc[i]=num\neeg_sub_id_num = eeg_sub_id_num.diff(-1)\neeg_sub_id_num = eeg_sub_id_num[eeg_sub_id_num['number_of_ids']>0]\n\nplt.bar(eeg_sub_id_num.index, eeg_sub_id_num['number_of_ids'], log=True)\nplt.ylabel('Frequency')\nplt.xlabel('Subsamples in one recording')\nplt.title(\"Bar of duplicated EEGs\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:25.412707Z","iopub.execute_input":"2024-01-14T10:37:25.413115Z","iopub.status.idle":"2024-01-14T10:37:27.071638Z","shell.execute_reply.started":"2024-01-14T10:37:25.413082Z","shell.execute_reply":"2024-01-14T10:37:27.069599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_sub_id_num","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:27.076864Z","iopub.execute_input":"2024-01-14T10:37:27.077559Z","iopub.status.idle":"2024-01-14T10:37:27.096739Z","shell.execute_reply.started":"2024-01-14T10:37:27.077506Z","shell.execute_reply":"2024-01-14T10:37:27.095079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### eeg_label_offset_seconds \n> **\"The time between the beginning of the consolidated EEG and this subsample.\"** - for clopping 50s eeg subset","metadata":{}},{"cell_type":"code","source":"# clop corresponding 50 second long subsample\ni = 2024\n\neeg1_id = train['eeg_id'][i] # == 6.0s\noffset_idx = int(train['eeg_label_offset_seconds'][i]*100) \n\neeg1 = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{eeg1_id}.parquet')\neeg1 = eeg1[offset_idx:offset_idx+5000] #slice 5000 slices = 50 sec.\nprint(f\"offset_sec = {train['eeg_label_offset_seconds'][i]}\")\neeg1","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:27.099299Z","iopub.execute_input":"2024-01-14T10:37:27.099773Z","iopub.status.idle":"2024-01-14T10:37:27.166726Z","shell.execute_reply.started":"2024-01-14T10:37:27.099737Z","shell.execute_reply":"2024-01-14T10:37:27.165351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### spectrogram_id\n>**\"A unique identifier for the entire EEG recording.\"** - Correspond to train_spectrograms","metadata":{}},{"cell_type":"code","source":"# load spectrogram by id\nspc0_id = train['spectrogram_id'][0]\nspc0 = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{spc0_id}.parquet')\nspc0","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:27.169771Z","iopub.execute_input":"2024-01-14T10:37:27.170168Z","iopub.status.idle":"2024-01-14T10:37:27.258491Z","shell.execute_reply.started":"2024-01-14T10:37:27.170137Z","shell.execute_reply":"2024-01-14T10:37:27.256256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### spectrogram_sub_id\n> **\"An ID for the specific 10 minute subsample this row's labels apply to.\"** - some train_spectrograms are recorded longer and More subsamples are extracted (same as eeg)","metadata":{}},{"cell_type":"code","source":"#number of subsamples per sample\nspctr_sub_id_num = pd.DataFrame(\n    np.zeros((len(train['spectrogram_sub_id'].unique()), 1)), columns=['number_of_ids']\n)\n\nfor i in range(len(train['spectrogram_sub_id'].unique())):\n    num = train[train['spectrogram_sub_id']==i].shape[0]\n    spctr_sub_id_num.iloc[i]=num\n\nspctr_sub_id_num = spctr_sub_id_num.diff(-1)\nspctr_sub_id_num = spctr_sub_id_num[spctr_sub_id_num['number_of_ids']>0]\n\nplt.bar(spctr_sub_id_num.index, spctr_sub_id_num['number_of_ids'], log=True)\nplt.ylabel('Frequency')\nplt.xlabel('Subsamples in one recording')\nplt.title(\"Bar of duplicated spectrograms\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:27.374312Z","iopub.execute_input":"2024-01-14T10:37:27.374774Z","iopub.status.idle":"2024-01-14T10:37:29.113198Z","shell.execute_reply.started":"2024-01-14T10:37:27.374740Z","shell.execute_reply":"2024-01-14T10:37:29.111478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### spectogram_label_offset_seconds \n> **\"The time between the beginning of the consolidated spectrogram and this subsample.\"** for clopping 10 min spectrogram subset","metadata":{}},{"cell_type":"code","source":"# clop corresponding 50 second long subsample\ni = 2024\nspctr_id = train['spectrogram_id'][i] # == 6.0s\noffset_idx = int(train['spectrogram_label_offset_seconds'][i]//2) \n\nspctr1 = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{spctr_id}.parquet')\nspctr1 = spctr1[offset_idx:offset_idx+300] #slice 300 slices = 10 min (0.5 slice/s) \n\nprint(f\"offset_sec = {train['spectrogram_label_offset_seconds'][i]}\")\nspctr1","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:30.911175Z","iopub.execute_input":"2024-01-14T10:37:30.911668Z","iopub.status.idle":"2024-01-14T10:37:30.997537Z","shell.execute_reply.started":"2024-01-14T10:37:30.911631Z","shell.execute_reply":"2024-01-14T10:37:30.996278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### labels_id\n> **\"An ID for this set of labels.\"** - 106800 unique values == rows of train","metadata":{}},{"cell_type":"markdown","source":"### patient_id\n> **\"An ID for the patient who donated the data.\"** - 1950 unique values -> several eegs and spectrograms are recorded","metadata":{}},{"cell_type":"code","source":"gb = train[['eeg_id', 'spectrogram_id', 'patient_id', 'label_id']].groupby('patient_id').nunique()\nplt.hist(gb['eeg_id'].sort_values().reset_index(drop=True), 30, range=(0,60),alpha=0.7, label=\"eeg\")\nplt.hist(gb['spectrogram_id'].sort_values().reset_index(drop=True), 30, range=(0,60),alpha=0.35, label=\"spectrogram\")\nplt.hist(gb['label_id'].sort_values().reset_index(drop=True), 30, range=(0,100),alpha=0.25, label=\"label\")\nplt.legend()\nplt.title('total labels per patient')\nplt.xlabel('experiment (subsamples)')\nplt.ylabel(\"patient\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:32.278777Z","iopub.execute_input":"2024-01-14T10:37:32.279362Z","iopub.status.idle":"2024-01-14T10:37:32.856509Z","shell.execute_reply.started":"2024-01-14T10:37:32.279300Z","shell.execute_reply":"2024-01-14T10:37:32.855395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gb = train[['patient_id', 'expert_consensus']].groupby('patient_id').nunique()\nfor i in np.sort(gb['expert_consensus'].unique()):\n    print(f'Number of consensus obtained in a patient:{i} | {len(gb[gb[\"expert_consensus\"]==i])} people')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:32.858575Z","iopub.execute_input":"2024-01-14T10:37:32.859534Z","iopub.status.idle":"2024-01-14T10:37:32.898288Z","shell.execute_reply.started":"2024-01-14T10:37:32.859488Z","shell.execute_reply":"2024-01-14T10:37:32.897132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### expert_consensus\n> **\"The consensus annotator label. Provided for convenience only.\"**","metadata":{}},{"cell_type":"code","source":"plt.figure()\nplt.title('Number of samples in each consensus')\nplt.xticks(np.arange(6)+1,train['expert_consensus'].unique())\nplt.bar(np.arange(6)+1.2, train[['label_id','expert_consensus']].groupby('expert_consensus').nunique()['label_id'],width=0.4,\n        alpha=0.5, color='C0', label='labels')\nplt.legend()\nplt.twinx()\nplt.bar(np.arange(6)+0.8, train[['patient_id','expert_consensus']].groupby('expert_consensus').nunique()['patient_id'],width=0.4,\n        alpha=0.5, color='C1', label='patients')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:37.115123Z","iopub.execute_input":"2024-01-14T10:37:37.115549Z","iopub.status.idle":"2024-01-14T10:37:37.642057Z","shell.execute_reply.started":"2024-01-14T10:37:37.115517Z","shell.execute_reply":"2024-01-14T10:37:37.640755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EEG and Spectrogram","metadata":{}},{"cell_type":"markdown","source":"### Subsamples","metadata":{}},{"cell_type":"code","source":"train_eegs_paths = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*.parquet')\ntrain_spectrograms_paths = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*.parquet')\n\nprint(f'EEG subsamples: {len(train_eegs_paths)} | spectrogram subsamples: {len(train_spectrograms_paths)}')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:37:39.704013Z","iopub.execute_input":"2024-01-14T10:37:39.704524Z","iopub.status.idle":"2024-01-14T10:37:39.836024Z","shell.execute_reply.started":"2024-01-14T10:37:39.704485Z","shell.execute_reply":"2024-01-14T10:37:39.835102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### seizure / lpd / gpd / lrda / grda / other_vote\n> **\"The count of annotator votes for a given brain activity class.** Explanation by Bard:\n#### EEA\n> - **lpd: lateralized periodic discharges** - Recurring every 0.5-3 seconds, focused on one side of the brain with sharp waves or spikes.\n<br>\n> - **gpd: generalized periodic discharges**  - Firing on both sides of the brain every 0.5-3 seconds. Unlike their \"lateralized\" cousins, they don't favor one side over the other.\n<br>\n> - **lrda: lateralized rhythmic delta activity** - A slow, rhythmic drumbeat focused on one side. 0.25-1 seconds.\n<br>\n> - **grda: generalized rhythmic delta activity** - Samely, Rhythmic drumbeats every 0.25-1 seconds, echoing throughout the brain.\n<br>\n> - **seizure** - Focal or generalized. More complex, organized, and often faster rhythmic activity compared to the isolated bursts of GPDs or the slow, repetitive waves of GRDA.\n\n#### Spectrogram\n\n> - **lpd: lateralized periodic discharges** - Isolated sharp spikes at regular intervals (0.5-3 seconds) within a specific band (usually higher than GPDs) appear like intermittent lightning flashes.\n<br>\n> - **gpd: generalized periodic discharges**  - Repetitive bursts of higher frequency spikes (0.5-3 seconds) across a broader band paint a picture of scattered rainstorms across the brain.\n<br>\n> - **lrda: lateralized rhythmic delta activity** - A sustained, rhythmic band of slow delta waves (1-4 Hz) dominates one side of the spectrum, like a droning bass line focused on one speaker.\n<br>\n> - **grda: generalized rhythmic delta activity** - A widespread, rhythmic band of slow delta waves (1-4 Hz) fills the entire spectrum, resembling a thick fog enveloping the whole soundscape.\n<br>\n> - **seizure** - Focal or generalized. More complex, organized, and often faster rhythmic activity compared to the isolated bursts of GPDs or the slow, repetitive waves of GRDA.","metadata":{}},{"cell_type":"markdown","source":"We would still like to see a specific example. Let's draw some obvious waves with constant higher votes in certain category.","metadata":{}},{"cell_type":"markdown","source":"### Seizure","metadata":{}},{"cell_type":"code","source":"seizure_sample = train.sort_values('seizure_vote',ascending=False).reset_index()\n\nseizure0_eeg_id = seizure_sample['eeg_id'][0]\nseizure0_spctr_id = seizure_sample['spectrogram_id'][0]\neeg0_offset_idx = int(seizure_sample['eeg_label_offset_seconds'][0]*100)\nspctr0_offset_idx = int(seizure_sample['spectrogram_label_offset_seconds'][0]//2) \n\nseizure0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{seizure0_eeg_id}.parquet')\nseizure0_eeg = seizure0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] #slice 5000 slices = 50 sec.\n\nseizure0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{seizure0_spctr_id}.parquet')\nseizure0_spctr = seizure0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300] #slice 300 slices = 10 min (0.5 slice/s) \n\nshow_eeg_spctr(seizure0_eeg, seizure0_spctr, title='Seizure')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T10:38:05.126971Z","iopub.execute_input":"2024-01-14T10:38:05.127412Z","iopub.status.idle":"2024-01-14T10:38:09.526562Z","shell.execute_reply.started":"2024-01-14T10:38:05.127376Z","shell.execute_reply":"2024-01-14T10:38:09.524633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LPD (Lateralized Periodic Discharges)","metadata":{}},{"cell_type":"code","source":"lpd_sample = train.sort_values('lpd_vote',ascending=False).reset_index()\n\nlpd0_eeg_id = lpd_sample['eeg_id'][0]\nlpd0_spctr_id = lpd_sample['spectrogram_id'][0]\neeg0_offset_idx = int(lpd_sample['eeg_label_offset_seconds'][0]*100)\nspctr0_offset_idx = int(lpd_sample['spectrogram_label_offset_seconds'][0]//2) \n\nlpd0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{lpd0_eeg_id}.parquet')\nlpd0_eeg = lpd0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] \n\nlpd0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{lpd0_spctr_id}.parquet')\nlpd0_spctr = lpd0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300]\n\nshow_eeg_spctr(lpd0_eeg, lpd0_spctr, title='LPD')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T09:53:33.235807Z","iopub.execute_input":"2024-01-14T09:53:33.236128Z","iopub.status.idle":"2024-01-14T09:53:35.770419Z","shell.execute_reply.started":"2024-01-14T09:53:33.236098Z","shell.execute_reply":"2024-01-14T09:53:35.769184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GPD (generalized periodic discharges)","metadata":{}},{"cell_type":"code","source":"gpd_sample = train.sort_values('gpd_vote',ascending=False).reset_index()\n\ngpd0_eeg_id = gpd_sample['eeg_id'][0]\ngpd0_spctr_id = gpd_sample['spectrogram_id'][0]\neeg0_offset_idx = int(gpd_sample['eeg_label_offset_seconds'][0]*100)\nspctr0_offset_idx = int(gpd_sample['spectrogram_label_offset_seconds'][0]//2) \n\ngpd0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{gpd0_eeg_id}.parquet')\ngpd0_eeg = gpd0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] \n\ngpd0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{gpd0_spctr_id}.parquet')\ngpd0_spctr = gpd0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300]\n\nshow_eeg_spctr(gpd0_eeg, gpd0_spctr, title='GPD')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T09:53:35.772435Z","iopub.execute_input":"2024-01-14T09:53:35.773636Z","iopub.status.idle":"2024-01-14T09:53:38.861553Z","shell.execute_reply.started":"2024-01-14T09:53:35.773588Z","shell.execute_reply":"2024-01-14T09:53:38.860311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LRDA (lateralized rhythmic delta activity)","metadata":{}},{"cell_type":"code","source":"lrda_sample = train.sort_values('lrda_vote',ascending=False).reset_index()\n\nlrda0_eeg_id = lrda_sample['eeg_id'][1]\nlrda0_spctr_id = lrda_sample['spectrogram_id'][1]\neeg0_offset_idx = int(lrda_sample['eeg_label_offset_seconds'][1]*100)\nspctr0_offset_idx = int(lrda_sample['spectrogram_label_offset_seconds'][1]//2) \n\nlrda0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{lrda0_eeg_id}.parquet')\nlrda0_eeg = lrda0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] \n\nlrda0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{lrda0_spctr_id}.parquet')\nlrda0_spctr = lrda0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300]\n\nshow_eeg_spctr(lrda0_eeg, lrda0_spctr, title='LRDA')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T09:53:38.863323Z","iopub.execute_input":"2024-01-14T09:53:38.863724Z","iopub.status.idle":"2024-01-14T09:53:42.673846Z","shell.execute_reply.started":"2024-01-14T09:53:38.863691Z","shell.execute_reply":"2024-01-14T09:53:42.670930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GRDA (generalized rhythmic delta activity)","metadata":{}},{"cell_type":"code","source":"grda_sample = train.sort_values('grda_vote',ascending=False).reset_index()\n\ngrda0_eeg_id = grda_sample['eeg_id'][1]\ngrda0_spctr_id = grda_sample['spectrogram_id'][1]\neeg0_offset_idx = int(grda_sample['eeg_label_offset_seconds'][1]*100)\nspctr0_offset_idx = int(grda_sample['spectrogram_label_offset_seconds'][1]//2) \n\ngrda0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{grda0_eeg_id}.parquet')\ngrda0_eeg = grda0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] \n\ngrda0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{grda0_spctr_id}.parquet')\ngrda0_spctr = grda0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300]\n\nshow_eeg_spctr(grda0_eeg, grda0_spctr, title='GRDA')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T09:53:42.675642Z","iopub.execute_input":"2024-01-14T09:53:42.676048Z","iopub.status.idle":"2024-01-14T09:53:46.159554Z","shell.execute_reply.started":"2024-01-14T09:53:42.676012Z","shell.execute_reply":"2024-01-14T09:53:46.157693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Other - Normal?","metadata":{}},{"cell_type":"code","source":"other_sample = train.sort_values('other_vote',ascending=False).reset_index()\n\nother0_eeg_id = other_sample['eeg_id'][0]\nother0_spctr_id = other_sample['spectrogram_id'][0]\neeg0_offset_idx = int(other_sample['eeg_label_offset_seconds'][0]*100)\nspctr0_offset_idx = int(other_sample['spectrogram_label_offset_seconds'][0]//2) \n\nother0_eeg = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{other0_eeg_id}.parquet')\nother0_eeg = other0_eeg[eeg0_offset_idx:eeg0_offset_idx+5000] #slice 5000 slices = 50 sec.\n\nother0_spctr = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{other0_spctr_id}.parquet')\nother0_spctr = other0_spctr.iloc[spctr0_offset_idx:spctr0_offset_idx+300] #slice 300 slices = 10 min (0.5 slice/s) \n\nshow_eeg_spctr(other0_eeg, other0_spctr, title='Other')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T09:53:46.162235Z","iopub.execute_input":"2024-01-14T09:53:46.162636Z","iopub.status.idle":"2024-01-14T09:53:49.308971Z","shell.execute_reply.started":"2024-01-14T09:53:46.162595Z","shell.execute_reply":"2024-01-14T09:53:49.307857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- In LRDA and GRDA, spikes occur in a constant rhythm over a short period of time, while LPD and GPD are slower rhythms.\n- The seizure shows more dynamic wave behavior.\n- Spectrogram intensity is consistently low in GPD and LPD.","metadata":{}},{"cell_type":"markdown","source":"### To Be Continued","metadata":{}},{"cell_type":"markdown","source":"- Baseline: Mean and std may work well??\n\n\nI will continue to update this notebook for my EDA. If you have any suggestions, please comment!","metadata":{}}]}