{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:12:07.085235Z","iopub.execute_input":"2024-04-07T20:12:07.085636Z","iopub.status.idle":"2024-04-07T20:12:20.179723Z","shell.execute_reply.started":"2024-04-07T20:12:07.085607Z","shell.execute_reply":"2024-04-07T20:12:20.178373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll create a new dataset of EEG spectrograms following the steps taken in [Christ Deotte's How to Make Spectrograms from EEG notebook](https://www.kaggle.com/code/cdeotte/how-to-make-spectrogram-from-eeg).","metadata":{}},{"cell_type":"markdown","source":"## Look at the Data","metadata":{}},{"cell_type":"code","source":"eeg_df = pd.read_parquet((path/'train_eegs').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:12:45.342210Z","iopub.execute_input":"2024-04-07T20:12:45.342613Z","iopub.status.idle":"2024-04-07T20:12:45.695260Z","shell.execute_reply.started":"2024-04-07T20:12:45.342571Z","shell.execute_reply":"2024-04-07T20:12:45.694355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:12:52.280924Z","iopub.execute_input":"2024-04-07T20:12:52.281572Z","iopub.status.idle":"2024-04-07T20:12:52.308399Z","shell.execute_reply.started":"2024-04-07T20:12:52.281537Z","shell.execute_reply":"2024-04-07T20:12:52.307614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:13:10.697252Z","iopub.execute_input":"2024-04-07T20:13:10.698090Z","iopub.status.idle":"2024-04-07T20:13:10.704954Z","shell.execute_reply.started":"2024-04-07T20:13:10.698052Z","shell.execute_reply":"2024-04-07T20:13:10.703811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:13:59.290720Z","iopub.execute_input":"2024-04-07T20:13:59.291601Z","iopub.status.idle":"2024-04-07T20:13:59.363660Z","shell.execute_reply.started":"2024-04-07T20:13:59.291565Z","shell.execute_reply":"2024-04-07T20:13:59.362362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Setup constants---this is directly taken from the reference [notebook](https://www.kaggle.com/code/cdeotte/how-to-make-spectrogram-from-eeg)","metadata":{}},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\ndirectory_path = '/kaggle/working/EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:16:01.580139Z","iopub.execute_input":"2024-04-07T20:16:01.581569Z","iopub.status.idle":"2024-04-07T20:16:01.587812Z","shell.execute_reply.started":"2024-04-07T20:16:01.581530Z","shell.execute_reply":"2024-04-07T20:16:01.586619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"They take the middle 50 seconds of data since at inference time, the test EEG data will be 50 seconds long. The data is collected at a frequency of 200 samples per second. So the total length of data is 50 x 200 = 10000. As the total number of samples is 12000 (the number of rows in the data) the middle 10k is from 1000 to 11000.","metadata":{}},{"cell_type":"code","source":"middle = (len(eeg_df)-10_000)//2\nmiddle","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:18:07.230190Z","iopub.execute_input":"2024-04-07T20:18:07.230605Z","iopub.status.idle":"2024-04-07T20:18:07.236967Z","shell.execute_reply.started":"2024-04-07T20:18:07.230572Z","shell.execute_reply":"2024-04-07T20:18:07.235888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_df = eeg_df.iloc[middle:middle+10_000]","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:18:55.513140Z","iopub.execute_input":"2024-04-07T20:18:55.513527Z","iopub.status.idle":"2024-04-07T20:18:55.518930Z","shell.execute_reply.started":"2024-04-07T20:18:55.513499Z","shell.execute_reply":"2024-04-07T20:18:55.517613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:18:59.393326Z","iopub.execute_input":"2024-04-07T20:18:59.393702Z","iopub.status.idle":"2024-04-07T20:18:59.400887Z","shell.execute_reply.started":"2024-04-07T20:18:59.393675Z","shell.execute_reply":"2024-04-07T20:18:59.399526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VARIABLE TO HOLD SPECTROGRAM\nimg = np.zeros((128,256,4),dtype='float32')","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:19:28.569187Z","iopub.execute_input":"2024-04-07T20:19:28.569575Z","iopub.status.idle":"2024-04-07T20:19:28.574509Z","shell.execute_reply.started":"2024-04-07T20:19:28.569546Z","shell.execute_reply":"2024-04-07T20:19:28.573623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The formula in the reference notebook, to go from EEG signals to the four brain regions (LL, LP, RP, RR) follows the pattern:\n\n```\nLL Spec = ( spec(Fp1 - F7) + spec(F7 - T3) + spec(T3 - T5) + spec(T5 - O1) )/4.\n```","metadata":{}},{"cell_type":"code","source":"FEATS","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:20:23.224564Z","iopub.execute_input":"2024-04-07T20:20:23.224951Z","iopub.status.idle":"2024-04-07T20:20:23.233015Z","shell.execute_reply.started":"2024-04-07T20:20:23.224920Z","shell.execute_reply":"2024-04-07T20:20:23.231951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll create the `LL` spectrogram by hand before going through the loop for all four regions:","metadata":{}},{"cell_type":"code","source":"k = 0","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:23:26.388137Z","iopub.execute_input":"2024-04-07T20:23:26.388860Z","iopub.status.idle":"2024-04-07T20:23:26.392721Z","shell.execute_reply.started":"2024-04-07T20:23:26.388826Z","shell.execute_reply":"2024-04-07T20:23:26.391903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COLS = FEATS[k]\nCOLS","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:23:26.805126Z","iopub.execute_input":"2024-04-07T20:23:26.806117Z","iopub.status.idle":"2024-04-07T20:23:26.812465Z","shell.execute_reply.started":"2024-04-07T20:23:26.806083Z","shell.execute_reply":"2024-04-07T20:23:26.811070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1 = eeg_df[COLS[0]].values - eeg_df[COLS[1]].values ","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:24:20.221891Z","iopub.execute_input":"2024-04-07T20:24:20.222685Z","iopub.status.idle":"2024-04-07T20:24:20.227339Z","shell.execute_reply.started":"2024-04-07T20:24:20.222651Z","shell.execute_reply":"2024-04-07T20:24:20.226401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:24:24.803152Z","iopub.execute_input":"2024-04-07T20:24:24.803786Z","iopub.status.idle":"2024-04-07T20:24:24.809050Z","shell.execute_reply.started":"2024-04-07T20:24:24.803753Z","shell.execute_reply":"2024-04-07T20:24:24.808185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FILL NANS\nm = np.nanmean(x1)\nm","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:25:09.662733Z","iopub.execute_input":"2024-04-07T20:25:09.663089Z","iopub.status.idle":"2024-04-07T20:25:09.670663Z","shell.execute_reply.started":"2024-04-07T20:25:09.663062Z","shell.execute_reply":"2024-04-07T20:25:09.669495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.isnan(x1).sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:25:50.302766Z","iopub.execute_input":"2024-04-07T20:25:50.303118Z","iopub.status.idle":"2024-04-07T20:25:50.311543Z","shell.execute_reply.started":"2024-04-07T20:25:50.303092Z","shell.execute_reply":"2024-04-07T20:25:50.310337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if entire x1 is not NaN, set NaN values to x1 mean, otherwise set x1 to 0\nif np.isnan(x1).mean()<1: \n    x1 = np.nan_to_num(x1,nan=m)\nelse: \n    x1[:] = 0","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:27:41.211421Z","iopub.execute_input":"2024-04-07T20:27:41.211804Z","iopub.status.idle":"2024-04-07T20:27:41.217555Z","shell.execute_reply.started":"2024-04-07T20:27:41.211773Z","shell.execute_reply":"2024-04-07T20:27:41.216440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1.mean()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:27:48.705748Z","iopub.execute_input":"2024-04-07T20:27:48.706910Z","iopub.status.idle":"2024-04-07T20:27:48.713066Z","shell.execute_reply.started":"2024-04-07T20:27:48.706874Z","shell.execute_reply":"2024-04-07T20:27:48.711969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n# RAW SPECTROGRAM\nmel_spec = librosa.feature.melspectrogram(y=x1, sr=200, hop_length=len(x1)//256, \n                                          n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:30:05.234126Z","iopub.execute_input":"2024-04-07T20:30:05.234630Z","iopub.status.idle":"2024-04-07T20:30:18.460812Z","shell.execute_reply.started":"2024-04-07T20:30:05.234598Z","shell.execute_reply":"2024-04-07T20:30:18.459441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:30:51.215035Z","iopub.execute_input":"2024-04-07T20:30:51.215439Z","iopub.status.idle":"2024-04-07T20:30:51.223335Z","shell.execute_reply.started":"2024-04-07T20:30:51.215404Z","shell.execute_reply":"2024-04-07T20:30:51.222251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOG TRANSFORM\nwidth = (mel_spec.shape[1]//32)*32\nwidth","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:34:16.348109Z","iopub.execute_input":"2024-04-07T20:34:16.348516Z","iopub.status.idle":"2024-04-07T20:34:16.355749Z","shell.execute_reply.started":"2024-04-07T20:34:16.348484Z","shell.execute_reply":"2024-04-07T20:34:16.354687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:37:16.388775Z","iopub.execute_input":"2024-04-07T20:37:16.389133Z","iopub.status.idle":"2024-04-07T20:37:16.397367Z","shell.execute_reply.started":"2024-04-07T20:37:16.389103Z","shell.execute_reply":"2024-04-07T20:37:16.396247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the librosa [docs](https://librosa.org/doc/0.10.1/generated/librosa.power_to_db.html#librosa-power-to-db), `power_to_db`:\n\n>Convert a power spectrogram (amplitude squared) to decibel (dB) units\n>  \n>This computes the scaling 10 * log10(S / ref) in a numerically stable way.","metadata":{}},{"cell_type":"code","source":"librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:37:09.270509Z","iopub.execute_input":"2024-04-07T20:37:09.270932Z","iopub.status.idle":"2024-04-07T20:37:09.280306Z","shell.execute_reply.started":"2024-04-07T20:37:09.270899Z","shell.execute_reply":"2024-04-07T20:37:09.279218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:50:01.734009Z","iopub.execute_input":"2024-04-07T20:50:01.734884Z","iopub.status.idle":"2024-04-07T20:50:01.740629Z","shell.execute_reply.started":"2024-04-07T20:50:01.734844Z","shell.execute_reply":"2024-04-07T20:50:01.739594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The min and max are `0.0` and `-80.0`, respectively.","metadata":{}},{"cell_type":"code","source":"mel_spec_db.max(), mel_spec_db.min()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:50:31.361888Z","iopub.execute_input":"2024-04-07T20:50:31.362728Z","iopub.status.idle":"2024-04-07T20:50:31.370093Z","shell.execute_reply.started":"2024-04-07T20:50:31.362685Z","shell.execute_reply":"2024-04-07T20:50:31.368957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# STANDARDIZE TO -1 TO 1\nmel_spec_db = (mel_spec_db+40)/40 ","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:51:03.761226Z","iopub.execute_input":"2024-04-07T20:51:03.761875Z","iopub.status.idle":"2024-04-07T20:51:03.766524Z","shell.execute_reply.started":"2024-04-07T20:51:03.761842Z","shell.execute_reply":"2024-04-07T20:51:03.764949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec_db.max(), mel_spec_db.min()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:51:09.381948Z","iopub.execute_input":"2024-04-07T20:51:09.382333Z","iopub.status.idle":"2024-04-07T20:51:09.389523Z","shell.execute_reply.started":"2024-04-07T20:51:09.382305Z","shell.execute_reply":"2024-04-07T20:51:09.388431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Store data into the placeholder `img`:","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:45:12.561377Z","iopub.execute_input":"2024-04-07T20:45:12.562140Z","iopub.status.idle":"2024-04-07T20:45:12.567557Z","shell.execute_reply.started":"2024-04-07T20:45:12.562103Z","shell.execute_reply":"2024-04-07T20:45:12.566213Z"}}},{"cell_type":"code","source":"mel_spec_db.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:52:04.210165Z","iopub.execute_input":"2024-04-07T20:52:04.210563Z","iopub.status.idle":"2024-04-07T20:52:04.218566Z","shell.execute_reply.started":"2024-04-07T20:52:04.210533Z","shell.execute_reply":"2024-04-07T20:52:04.217180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img[:,:,k] += mel_spec_db","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:51:45.130716Z","iopub.execute_input":"2024-04-07T20:51:45.131370Z","iopub.status.idle":"2024-04-07T20:51:45.136960Z","shell.execute_reply.started":"2024-04-07T20:51:45.131315Z","shell.execute_reply":"2024-04-07T20:51:45.135769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T20:52:08.926192Z","iopub.execute_input":"2024-04-07T20:52:08.926583Z","iopub.status.idle":"2024-04-07T20:52:08.933478Z","shell.execute_reply.started":"2024-04-07T20:52:08.926543Z","shell.execute_reply":"2024-04-07T20:52:08.932041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that I mostly understand how each spectrogram is created, I'll run through the reference notebook's for-loop, which generates all four regions' spectrograms.","metadata":{}},{"cell_type":"code","source":"img = np.zeros((128,256,4),dtype='float32')","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:38:01.836666Z","iopub.execute_input":"2024-04-07T21:38:01.837286Z","iopub.status.idle":"2024-04-07T21:38:01.841872Z","shell.execute_reply.started":"2024-04-07T21:38:01.837254Z","shell.execute_reply":"2024-04-07T21:38:01.840665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg_df[COLS[kk]].values - eeg_df[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:38:02.526694Z","iopub.execute_input":"2024-04-07T21:38:02.527365Z","iopub.status.idle":"2024-04-07T21:38:02.712285Z","shell.execute_reply.started":"2024-04-07T21:38:02.527320Z","shell.execute_reply":"2024-04-07T21:38:02.711078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking that `img` contains values:","metadata":{}},{"cell_type":"code","source":"img[:,:,0].mean(), \\\nimg[:,:,1].mean(), \\\nimg[:,:,2].mean(), \\\nimg[:,:,3].mean(),","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:38:04.522878Z","iopub.execute_input":"2024-04-07T21:38:04.523291Z","iopub.status.idle":"2024-04-07T21:38:04.533222Z","shell.execute_reply.started":"2024-04-07T21:38:04.523259Z","shell.execute_reply":"2024-04-07T21:38:04.532036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here's what one of those regions looks like","metadata":{}},{"cell_type":"code","source":"im = PILImage.create(Image.fromarray((img[:,:,3]* 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:39:10.802026Z","iopub.execute_input":"2024-04-07T21:39:10.803026Z","iopub.status.idle":"2024-04-07T21:39:10.815098Z","shell.execute_reply.started":"2024-04-07T21:39:10.802980Z","shell.execute_reply":"2024-04-07T21:39:10.814010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next I'll stack the four regions:","metadata":{}},{"cell_type":"code","source":"arrays = [img[:, :, i] for i in range(4)]","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:39:42.068298Z","iopub.execute_input":"2024-04-07T21:39:42.068714Z","iopub.status.idle":"2024-04-07T21:39:42.074193Z","shell.execute_reply.started":"2024-04-07T21:39:42.068682Z","shell.execute_reply":"2024-04-07T21:39:42.073405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = np.concatenate(arrays)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:42:20.659908Z","iopub.execute_input":"2024-04-07T21:42:20.660393Z","iopub.status.idle":"2024-04-07T21:42:20.665562Z","shell.execute_reply.started":"2024-04-07T21:42:20.660357Z","shell.execute_reply":"2024-04-07T21:42:20.664419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:42:23.506651Z","iopub.execute_input":"2024-04-07T21:42:23.507080Z","iopub.status.idle":"2024-04-07T21:42:23.513819Z","shell.execute_reply.started":"2024-04-07T21:42:23.507027Z","shell.execute_reply":"2024-04-07T21:42:23.512678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And visualize it:","metadata":{}},{"cell_type":"code","source":"im = PILImage.create(Image.fromarray((img* 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:42:26.510785Z","iopub.execute_input":"2024-04-07T21:42:26.511137Z","iopub.status.idle":"2024-04-07T21:42:26.536395Z","shell.execute_reply.started":"2024-04-07T21:42:26.511108Z","shell.execute_reply":"2024-04-07T21:42:26.535561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now I'll put those steps into the function provided in the reference notebook, and save it as a dataset. The steps I've added are stacking the 4 regions' spectrograms into one and saving as an image.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')\n    \ncols = ['eeg_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nagg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\nunique_df = df[cols].groupby(['eeg_id'], as_index=False).agg(agg_funcs)\nunique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n\nunique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:10:17.207167Z","iopub.execute_input":"2024-04-07T22:10:17.207582Z","iopub.status.idle":"2024-04-07T22:10:17.491477Z","shell.execute_reply.started":"2024-04-07T22:10:17.207552Z","shell.execute_reply":"2024-04-07T22:10:17.490410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:10:28.937748Z","iopub.execute_input":"2024-04-07T22:10:28.938144Z","iopub.status.idle":"2024-04-07T22:10:28.944316Z","shell.execute_reply.started":"2024-04-07T22:10:28.938113Z","shell.execute_reply":"2024-04-07T22:10:28.943113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def spectrogram_from_eeg(eeg_id):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(f'{PATH}{eeg_id}.parquet')\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n    img = np.concatenate([img[:, :, i] for i in range(4)])\n    \n    # read the label\n    label = unique_df[unique_df.eeg_id == eeg_id][\"target\"].item()\n    \n    # convert to image\n    im = PILImage.create(Image.fromarray((img * 255).astype(np.uint8)))\n    \n    # SAVE TO DISK\n    im.save(f\"{SPEC_DIR}/EEG_Spectrograms/{label}/{eeg_id}.png\")\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:50:53.622104Z","iopub.execute_input":"2024-04-07T21:50:53.622961Z","iopub.status.idle":"2024-04-07T21:50:53.635330Z","shell.execute_reply.started":"2024-04-07T21:50:53.622927Z","shell.execute_reply":"2024-04-07T21:50:53.634433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, I'll create folders for each image class:","metadata":{}},{"cell_type":"code","source":"# create output folders to hold spectrograms\nSPEC_DIR = \"/kaggle/working\"\nos.makedirs(SPEC_DIR+'/EEG_Spectrograms', exist_ok=True)\n\nfor targ in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    os.makedirs(SPEC_DIR+'/EEG_Spectrograms'+'/'+targ, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T21:46:58.971120Z","iopub.execute_input":"2024-04-07T21:46:58.971556Z","iopub.status.idle":"2024-04-07T21:46:58.977730Z","shell.execute_reply.started":"2024-04-07T21:46:58.971523Z","shell.execute_reply":"2024-04-07T21:46:58.976923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then I'll prepare the targets for each `eeg_id`:","metadata":{}},{"cell_type":"code","source":"import warnings","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_ids = unique_df.eeg_id.unique()\n\n\nwarnings.filterwarnings(\"ignore\")\nparallel(spectrogram_from_eeg, eeg_ids, n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path(\"/kaggle/working/EEG_Spectrograms\").ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:07:50.722509Z","iopub.execute_input":"2024-04-07T22:07:50.722916Z","iopub.status.idle":"2024-04-07T22:07:50.731625Z","shell.execute_reply.started":"2024-04-07T22:07:50.722883Z","shell.execute_reply":"2024-04-07T22:07:50.730434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll view a few images to make sure they look okay:","metadata":{}},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/EEG_Spectrograms/lrda_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:07:53.389322Z","iopub.execute_input":"2024-04-07T22:07:53.390232Z","iopub.status.idle":"2024-04-07T22:07:53.454507Z","shell.execute_reply.started":"2024-04-07T22:07:53.390177Z","shell.execute_reply":"2024-04-07T22:07:53.453412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/EEG_Spectrograms/gpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:07:55.701521Z","iopub.execute_input":"2024-04-07T22:07:55.701928Z","iopub.status.idle":"2024-04-07T22:07:55.766928Z","shell.execute_reply.started":"2024-04-07T22:07:55.701896Z","shell.execute_reply":"2024-04-07T22:07:55.766105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/EEG_Spectrograms/lpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:07:58.383495Z","iopub.execute_input":"2024-04-07T22:07:58.384484Z","iopub.status.idle":"2024-04-07T22:07:58.447725Z","shell.execute_reply.started":"2024-04-07T22:07:58.384439Z","shell.execute_reply":"2024-04-07T22:07:58.446665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/EEG_Spectrograms/other_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:07:59.222389Z","iopub.execute_input":"2024-04-07T22:07:59.223037Z","iopub.status.idle":"2024-04-07T22:07:59.286583Z","shell.execute_reply.started":"2024-04-07T22:07:59.222998Z","shell.execute_reply":"2024-04-07T22:07:59.285473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll spot-check the images to make sure that they are in the correct folders (class labels).","metadata":{}},{"cell_type":"code","source":"for vote in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    for fpath in (Path(\"/kaggle/working/EEG_Spectrograms\")/vote).ls()[:3]:\n        print(vote, vote == unique_df[unique_df.eeg_id == int(fpath.stem)]['target'].item())","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:08:16.409430Z","iopub.execute_input":"2024-04-07T22:08:16.409814Z","iopub.status.idle":"2024-04-07T22:08:16.438279Z","shell.execute_reply.started":"2024-04-07T22:08:16.409787Z","shell.execute_reply":"2024-04-07T22:08:16.437117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll make sure that I captured all of the images in the DataFrame:","metadata":{}},{"cell_type":"code","source":"files = get_image_files(Path(\"/kaggle/working/EEG_Spectrograms\"))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:08:22.384641Z","iopub.execute_input":"2024-04-07T22:08:22.384981Z","iopub.status.idle":"2024-04-07T22:08:22.416283Z","shell.execute_reply.started":"2024-04-07T22:08:22.384955Z","shell.execute_reply":"2024-04-07T22:08:22.415307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-07T22:08:23.808814Z","iopub.execute_input":"2024-04-07T22:08:23.809186Z","iopub.status.idle":"2024-04-07T22:08:23.815609Z","shell.execute_reply.started":"2024-04-07T22:08:23.809158Z","shell.execute_reply":"2024-04-07T22:08:23.814289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll save this notebook version and then convert the output folder to a Kaggle Dataset that I can use for training.","metadata":{}}]}