{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\n\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape', train.shape )\ndisplay( train.head() )","metadata":{"execution":{"iopub.status.busy":"2024-03-16T12:02:35.206149Z","iopub.execute_input":"2024-03-16T12:02:35.206630Z","iopub.status.idle":"2024-03-16T12:02:36.991927Z","shell.execute_reply.started":"2024-03-16T12:02:35.206588Z","shell.execute_reply":"2024-03-16T12:02:36.990530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2'],]\n\nFEATS_ADD = [['Fp1','F7','F3','C3','T3'],\n             ['Fp2','F8','F4','C4','T4'],\n             ['O1','T5','P3','C3','T3'],\n             ['O2','T6','P4','C4','T4'],]","metadata":{"execution":{"iopub.status.busy":"2024-03-16T12:02:36.993806Z","iopub.execute_input":"2024-03-16T12:02:36.994322Z","iopub.status.idle":"2024-03-16T12:02:37.004245Z","shell.execute_reply.started":"2024-03-16T12:02:36.994289Z","shell.execute_reply":"2024-03-16T12:02:37.001568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.ndimage import gaussian_filter\nfrom scipy.signal import butter, filtfilt, firwin\n\ndef total_filter(data, sampling_rate=200, order=2):\n    # Normalization\n    nyquist = 0.5 * sampling_rate\n    # First Filtering: Notch Filtering[특정 주파수 제거]\n    notch_freq = 60\n    Q = 30\n    notch_freq_normalized = notch_freq / nyquist\n    # b, a = butter(order, [notch_freq_normalized], btype='bandstop')\n    \n    # filtered_data = filtfilt(b, a, data, axis=0)\n    # Second Filtering: Band Pass Filtering[특정 주파수 강조] \n    # 강조할 주파수 대역: 0.5~20Hz \n    normal_lowest_cutoff = 0.5 / nyquist\n    normal_highest_cutoff = 20 / nyquist\n    num_taps = 100  # 필터의 길이, 적절한 값으로 조정해야 함\n    b = firwin(num_taps, [normal_lowest_cutoff, normal_highest_cutoff], pass_zero=False)\n    \n    filtered_data = filtfilt(b, 1, data, axis=0)\n    \n    # Third Filtering: Guassian Pass[Image Smoothing]\n    # Small Sigma: 저주파수 성분 보존, 고주파 잡음을 감소\n    # Big Sigma: 더 작은 지역적인 특징 \n    filtered_data = gaussian_filter(filtered_data, sigma=1, order=2, mode='reflect')\n    return filtered_data\n","metadata":{"execution":{"iopub.status.busy":"2024-03-16T12:02:37.005702Z","iopub.execute_input":"2024-03-16T12:02:37.006102Z","iopub.status.idle":"2024-03-16T12:02:38.351576Z","shell.execute_reply.started":"2024-03-16T12:02:37.006068Z","shell.execute_reply":"2024-03-16T12:02:38.350043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img1 = np.zeros((64,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        if k < 4:\n            for kk in range(4):\n        \n                # COMPUTE PAIR DIFFERENCES\n                x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n                # FILL NANS\n                m = np.nanmean(x)\n                if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n                else: x[:] = 0\n\n                total_filter(x)    \n                    \n                signals.append(x)\n\n                # RAW SPECTROGRAM\n                mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                     n_fft=1024, n_mels=64, fmin=0, fmax=20, win_length=64)\n\n                # LOG TRANSFORM\n                width = (mel_spec.shape[1]//32)*32\n                mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n                # STANDARDIZE TO -1 TO 1\n                mel_spec_db = (mel_spec_db+40)/40 \n                img1[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n            img1[:,:,k] /= 4.0\n            \n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img2 = np.zeros((64,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS_ADD[k]\n        if k < 4:\n            for kk in range(4):\n        \n                # COMPUTE PAIR DIFFERENCES\n                x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n                # FILL NANS\n                m = np.nanmean(x)\n                if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n                else: x[:] = 0\n\n                total_filter(x)    \n                    \n                signals.append(x)\n\n                # RAW SPECTROGRAM\n                mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                     n_fft=1024, n_mels=64, fmin=0, fmax=20, win_length=64)\n\n                # LOG TRANSFORM\n                width = (mel_spec.shape[1]//32)*32\n                mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n                # STANDARDIZE TO -1 TO 1\n                mel_spec_db = (mel_spec_db+40)/40 \n                img2[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n            img2[:,:,k] /= 4.0        \n        \n    img = np.concatenate((img1, img2), axis=0)\n        \n    if display:\n        for k in range(4):\n            plt.subplot(3,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n        plt.show()\n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-16T12:02:38.354711Z","iopub.execute_input":"2024-03-16T12:02:38.355231Z","iopub.status.idle":"2024-03-16T12:02:38.392166Z","shell.execute_reply.started":"2024-03-16T12:02:38.355184Z","shell.execute_reply":"2024-03-16T12:02:38.390506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\nEEG_IDS = train.eeg_id.unique()\nall_eegs = {}\n\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='')\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH}{eeg_id}.parquet', i<DISPLAY)\n    \n    # SAVE TO DISK\n    if i==DISPLAY:\n        print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n    all_eegs[eeg_id] = img\n   \n# SAVE EEG SPECTROGRAM DICTIONARY\nnp.save('eeg_specs',all_eegs)","metadata":{"execution":{"iopub.status.busy":"2024-03-16T12:02:38.394266Z","iopub.execute_input":"2024-03-16T12:02:38.394682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}