{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EDA: Missing Data in HMS-HBA Competition Spectrograms\n\nWhile processing EEG from a thrid-party source and creating spectrograms, I compared with competition spectrograms to make sure the spctrograms were similar. During that process, I happened upon a competition spectrogram with significant missing data (`/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/9661509.parquet`). While I expected some missing data, this particular spectrogram was missing more than a third of the data for a complete 10 minute spectrogram (see below). I found this surprising, and wanted to investigate further. \n\nThe result were surprising and might affect training with competition-provided spectrograms:\n\n* 7.2% of the label_id have missing data in the offset spectrogram.\n* 8.7% of spectrograms have missing data in at least one of the offset spectrograms.\n\nSee discussion [here](https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/discussion/478233).","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-19T18:17:34.645049Z","iopub.execute_input":"2024-02-19T18:17:34.645577Z","iopub.status.idle":"2024-02-19T18:17:50.245104Z","shell.execute_reply.started":"2024-02-19T18:17:34.645528Z","shell.execute_reply":"2024-02-19T18:17:50.243747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/9661509.parquet')\n\n\nspec = spec.fillna(0)\nsig = spec.iloc[:300, 1:101].T.values\n\n# Log spectrogram \nsig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0)) # avoid 0 in log\nsig = tf.math.log(sig)\n\n# Normalize spectrogram\nsig -= tf.math.reduce_mean(sig)\nsig /= tf.math.reduce_std(sig) + 1e-6\n\n# Plot the spectrogram\ntimes = spec.iloc[:300]['time']\nfrequencies = [float(c.split('_')[-1]) for c in spec.columns if c[:2] == 'LL']\nimg = sig.numpy() \nimg -= img.min()\nimg /= img.max() + 1e-4\nplt.figure(figsize=(10, 4))\nplt.pcolormesh(times, frequencies, img, shading='gouraud')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.title('Linear-frequency power spectrogram')\n# plt.colorbar(format='%+2.0f dB')\nplt.tight_layout()\nplt.savefig('missing_data.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T18:49:15.654538Z","iopub.execute_input":"2024-02-19T18:49:15.655041Z","iopub.status.idle":"2024-02-19T18:49:16.984382Z","shell.execute_reply.started":"2024-02-19T18:49:15.655008Z","shell.execute_reply":"2024-02-19T18:49:16.983398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf['missing_data'] = 0.0","metadata":{"execution":{"iopub.status.busy":"2024-02-19T18:17:51.242296Z","iopub.execute_input":"2024-02-19T18:17:51.243152Z","iopub.status.idle":"2024-02-19T18:17:51.543325Z","shell.execute_reply.started":"2024-02-19T18:17:51.243110Z","shell.execute_reply":"2024-02-19T18:17:51.542094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load all spectrograms and offset, then calculate the percentage of missing data.","metadata":{}},{"cell_type":"code","source":"for spec_id, dff in df.groupby('spectrogram_id'):\n    spec = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/{spec_id}.parquet')\n    for idx, row in dff.iterrows():\n        offset = int(row['spectrogram_label_offset_seconds'] // 2)\n        df.loc[idx, 'missing_data'] = spec.iloc[offset: 300 + offset, 1:].isna().mean().mean()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T18:17:51.546626Z","iopub.execute_input":"2024-02-19T18:17:51.547136Z","iopub.status.idle":"2024-02-19T18:30:27.161779Z","shell.execute_reply.started":"2024-02-19T18:17:51.547081Z","shell.execute_reply":"2024-02-19T18:30:27.160281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{(df['missing_data'] > 0).mean() * 100: 0.1f}% of label_ids have missing data.\")\nprint(f\"{(df.groupby('spectrogram_id')['missing_data'].max()!=0).mean() * 100: 0.1f}% of spectrogram_ids have missing data.\")","metadata":{"execution":{"iopub.status.busy":"2024-02-19T18:38:52.355578Z","iopub.execute_input":"2024-02-19T18:38:52.356603Z","iopub.status.idle":"2024-02-19T18:38:52.370263Z","shell.execute_reply.started":"2024-02-19T18:38:52.356561Z","shell.execute_reply":"2024-02-19T18:38:52.368681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove spectrograms with no missing data\ndf_missing = df.loc[df['missing_data'] > 0, 'missing_data']\ndf_missing.plot.hist(bins=20, title='Missing Data Distribution by `label_id`', xlabel='Portion of Missing Data');\nplt.savefig('missing_data_label_id.png')","metadata":{"execution":{"iopub.status.busy":"2024-02-19T19:03:56.205623Z","iopub.execute_input":"2024-02-19T19:03:56.206152Z","iopub.status.idle":"2024-02-19T19:03:56.623256Z","shell.execute_reply.started":"2024-02-19T19:03:56.206114Z","shell.execute_reply":"2024-02-19T19:03:56.621858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Look at max missing data per spectrogram\ndf.loc[df['missing_data'] > 0].groupby('spectrogram_id')['missing_data'].max().plot.hist(bins=20, title='Max Missing Data Distribution by `spectrogram_id`', xlabel='Portion of Missing Data');\nplt.savefig('missing_data_spec_id')","metadata":{"execution":{"iopub.status.busy":"2024-02-19T19:03:33.248661Z","iopub.execute_input":"2024-02-19T19:03:33.249154Z","iopub.status.idle":"2024-02-19T19:03:33.640257Z","shell.execute_reply.started":"2024-02-19T19:03:33.249123Z","shell.execute_reply":"2024-02-19T19:03:33.638786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T18:30:27.767952Z","iopub.execute_input":"2024-02-19T18:30:27.768512Z","iopub.status.idle":"2024-02-19T18:30:28.849008Z","shell.execute_reply.started":"2024-02-19T18:30:27.768465Z","shell.execute_reply":"2024-02-19T18:30:28.847689Z"},"trusted":true},"execution_count":null,"outputs":[]}]}