{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom scipy import signal\nimport pickle\n\nTRAIN = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\nTEST = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'\n\nSG = '/kaggle/input/hms-harmful-brain-activity-classification/%s_spectrograms/%i.parquet'\nEEG = '/kaggle/input/hms-harmful-brain-activity-classification/%s_eegs/%i.parquet'\n\nEEG_FS = 200\nEEG_LEN = EEG_FS * 50\n\nSG_LEN = 300\n\ndf_train = pd.read_csv(TRAIN)\ndf_test = pd.read_csv(TEST)\ndf_train","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-15T10:05:14.025436Z","iopub.execute_input":"2024-01-15T10:05:14.025808Z","iopub.status.idle":"2024-01-15T10:05:15.898666Z","shell.execute_reply.started":"2024-01-15T10:05:14.025776Z","shell.execute_reply":"2024-01-15T10:05:15.897767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_item(item_id):\n    item = df_train.iloc[item_id]\n    eeg_start = int(EEG_FS * item.eeg_label_offset_seconds)\n    eeg = pd.read_parquet(EEG % ('train', item.eeg_id)).iloc[eeg_start:eeg_start + EEG_LEN].reset_index()\n    sg_start = int(item.spectrogram_label_offset_seconds / 2)\n    sg = pd.read_parquet(SG % ('train', item.spectrogram_id)).iloc[sg_start:sg_start + SG_LEN, 1:].reset_index()\n    return sg, eeg, item.iloc[-6:]\n    \ndef get_test_item(item_id):\n    item = df_test.iloc[item_id]\n    \n    eeg_start = 0\n    eeg = pd.read_parquet(EEG % ('test', item.eeg_id)).reset_index()\n    sg = pd.read_parquet(SG % ('test', item.spectrogram_id)).iloc[:, 1:].reset_index()\n    return sg, eeg, None\n\ndef get_train_item_small(item_id, t_eeg=10, t_sg=10, decimate_q=5):\n    \"\"\"This function selects only a small version of input data\"\"\"\n    item = df_train.iloc[item_id]\n    eeg_start = int(EEG_FS * item.eeg_label_offset_seconds + (EEG_LEN - t_eeg * EEG_FS) / 2)\n    eeg = pd.read_parquet(EEG % ('train', item.eeg_id)).iloc[eeg_start:eeg_start + t_eeg * EEG_FS].fillna(value=0)\n    eeg = signal.decimate(eeg, decimate_q, axis=0)\n    \n    sg_start = int(item.spectrogram_label_offset_seconds / 2 + (SG_LEN - t_sg)/2)\n    sg = pd.read_parquet(SG % ('train', item.spectrogram_id)).iloc[sg_start:sg_start + t_sg, 1:].to_numpy().reshape((-1, 4, 100))\n    sg = np.moveaxis(sg, 1, 0)\n    \n    return sg, eeg, item.iloc[-6:].to_numpy()\n\ndef view_sg(sg):\n    x = sg.to_numpy().T\n    names = ['LL', 'RL', 'LP', 'RP']\n    fig, ax = plt.subplots(ncols=3, nrows=4, figsize=(24, 16))\n    for n in range(4):\n        ax[n, 0].pcolormesh(np.arange(-150, 150), np.arange(0, 100), x[n*100:(n+1)*100])\n        ax[n, 1].pcolormesh(np.arange(-10, 10), np.arange(0, 100), x[n*100:(n+1)*100, 140:160])    \n        ax[n, 2].pcolormesh(np.arange(-5, 5), np.arange(0, 100), x[n*100:(n+1)*100, 145:155])\n\n        ax[n, 0].set_title(names[n])\n\ndef view_eeg(eeg):\n    butt_filt = signal.butter(3, [1e-5, 0.2], btype='bandpass')\n\n    names = [[('Fp1', 'F3'), ('F3', 'C3'), ('C3', 'P3'), ('P3', 'O1')], [('Fp2', 'F4'), ('F4', 'C4'), ('C4', 'P4'), ('P4', 'O2')],\n             [('Fp1', 'F7'), ('F7', 'T3'), ('T3', 'T5'), ('T5', 'O1')], [('Fp2', 'F8'), ('F8', 'T4'), ('T4', 'T6'), ('T6', 'O2')],\n             [('Fz', 'Cz'), ('Cz', 'Pz')]]\n\n    fig, ax = plt.subplots(ncols=2, nrows=6, figsize=(36, 24))\n    for n, els in enumerate(names):\n        for el1, el2 in els:\n            x = signal.filtfilt(*butt_filt, eeg[el1] - eeg[el2])\n            #x = eeg[el1] - eeg[el2]\n            \n            ax[n, 0].plot(np.arange(-25, 25, 1/EEG_FS), x, label=f'{el1}-{el2}')\n            ax[n, 0].legend(loc='lower left')\n\n            ax[n, 1].plot(np.arange(-5, 5, 1/EEG_FS), x[4000:6000], label=f'{el1}-{el2}')\n            ax[n, 1].legend(loc='lower left')\n    \n    x = signal.filtfilt(*butt_filt, eeg['EKG'])\n    ax[5, 0].plot(np.arange(-25, 25, 1/EEG_FS), x, label='EKG')\n    ax[5, 0].legend(loc='lower left')\n\n    ax[5, 1].plot(np.arange(-5, 5, 1/EEG_FS), x[4000:6000], label='EKG')\n    ax[5, 1].legend(loc='lower left')       \n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-15T10:05:15.900431Z","iopub.execute_input":"2024-01-15T10:05:15.901023Z","iopub.status.idle":"2024-01-15T10:05:15.923912Z","shell.execute_reply.started":"2024-01-15T10:05:15.900991Z","shell.execute_reply":"2024-01-15T10:05:15.920454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_id = 814\n\nsg, eeg, label = get_train_item(item_id)\n\nview_eeg(eeg)\nview_sg(sg)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T10:05:15.925263Z","iopub.execute_input":"2024-01-15T10:05:15.925612Z","iopub.status.idle":"2024-01-15T10:05:22.643623Z","shell.execute_reply.started":"2024-01-15T10:05:15.925581Z","shell.execute_reply":"2024-01-15T10:05:22.642520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dump a compact verison of all data","metadata":{}},{"cell_type":"code","source":"sg_data, eeg_data = [], []\nN = 10000\nfor n in range(len(df_train)):\n    sg, eeg, label = get_train_item_small(n, t_eeg=20, t_sg=20)\n    sg_data.append(sg)\n    eeg_data.append(eeg)\n\n    if len(sg_data) == N:\n        print(n + 1)\n        with open(f'sg_{n // N}.pkl', 'wb') as f:\n            pickle.dump(np.stack(sg_data), f)\n        with open(f'eeg_{n // N}.pkl', 'wb') as f:\n            pickle.dump(np.stack(eeg_data), f)\n        \n        sg_data, eeg_data = [], []\n    \nwith open(f'sg_{n // N}.pkl', 'wb') as f:\n    pickle.dump(np.stack(sg_data), f)\nwith open(f'eeg_{n // N}.pkl', 'wb') as f:\n    pickle.dump(np.stack(eeg_data), f)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T10:15:04.716245Z","iopub.execute_input":"2024-01-15T10:15:04.716628Z","iopub.status.idle":"2024-01-15T10:15:18.530201Z","shell.execute_reply.started":"2024-01-15T10:15:04.716595Z","shell.execute_reply":"2024-01-15T10:15:18.528767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}