{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":171463847,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ariyasu","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport warnings\nwarnings.simplefilter('ignore')\npd.set_option('display.max_columns', 100)\nimport cv2\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nfrom scipy.special import softmax\nimport multiprocessing\nimport copy\n\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom tqdm import tqdm\nimport random\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:30.756149Z","iopub.execute_input":"2024-04-11T07:26:30.756716Z","iopub.status.idle":"2024-04-11T07:26:30.766355Z","shell.execute_reply.started":"2024-04-11T07:26:30.756678Z","shell.execute_reply":"2024-04-11T07:26:30.764592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there is not enough storage, create a spectrogram divided by 5 notebooks.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/preprocess-train-csv/train_with_group_in_eeg.csv').sort_values('eeg_id')\ndf['eeg_path'] = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eeg/'+df.eeg_id.astype(str)+'.parquet'\ndf['spe_path'] = f'/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'+df.spectrogram_id.astype(str)+'.parquet'\ndf.eeg_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:30.771213Z","iopub.execute_input":"2024-04-11T07:26:30.772321Z","iopub.status.idle":"2024-04-11T07:26:31.560078Z","shell.execute_reply.started":"2024-04-11T07:26:30.772264Z","shell.execute_reply":"2024-04-11T07:26:31.558911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_n = 5\nthis_note_data_n = 3","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:31.562806Z","iopub.execute_input":"2024-04-11T07:26:31.563872Z","iopub.status.idle":"2024-04-11T07:26:31.569358Z","shell.execute_reply.started":"2024-04-11T07:26:31.563813Z","shell.execute_reply":"2024-04-11T07:26:31.567896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_eeg_ids = df.eeg_id.unique()\nchunk = len(all_eeg_ids) // all_data_n + 1\nuse_eeg_ids = all_eeg_ids[this_note_data_n*chunk:(this_note_data_n+1)*chunk]\nlen(use_eeg_ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:31.570826Z","iopub.execute_input":"2024-04-11T07:26:31.571384Z","iopub.status.idle":"2024-04-11T07:26:31.587576Z","shell.execute_reply.started":"2024-04-11T07:26:31.571351Z","shell.execute_reply":"2024-04-11T07:26:31.585407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df.eeg_id.isin(use_eeg_ids)]\ndf","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:31.589641Z","iopub.execute_input":"2024-04-11T07:26:31.59085Z","iopub.status.idle":"2024-04-11T07:26:31.644326Z","shell.execute_reply.started":"2024-04-11T07:26:31.5908Z","shell.execute_reply":"2024-04-11T07:26:31.643084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# egg to spectrogram","metadata":{}},{"cell_type":"code","source":"PATH = f'/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nEEG_IDS = df.eeg_id.unique()\nUSE_WAVELET = None\nDISPLAY = 0","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:31.647908Z","iopub.execute_input":"2024-04-11T07:26:31.649009Z","iopub.status.idle":"2024-04-11T07:26:31.656991Z","shell.execute_reply.started":"2024-04-11T07:26:31.648956Z","shell.execute_reply":"2024-04-11T07:26:31.655618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nfrom tqdm import tqdm\nimport pywt\nimport multiprocessing\nfrom scipy.signal import butter, lfilter\n\ndef butter_bandpass(lowcut, highcut, fs, order=2):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype=\"band\")\n    return b, a\n\n\ndef butter_bandpass_filter(data, lowcut=0.5, highcut=30, fs=200, order=2):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:31.658929Z","iopub.execute_input":"2024-04-11T07:26:31.659723Z","iopub.status.idle":"2024-04-11T07:26:32.164636Z","shell.execute_reply.started":"2024-04-11T07:26:31.659684Z","shell.execute_reply":"2024-04-11T07:26:32.163656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\ndef spectrogram_from_eeg(args):\n    a, fmax, win_length, hop, w_cut, display, sec, apply_butter_bandpass_filter = args\n    key, idf = a\n    eeg_id, eeg_group_num = key\n    parquet_path = f'{PATH}{eeg_id}.parquet'\n    middle_sec = int((idf.eeg_label_offset_seconds.max() + idf.eeg_label_offset_seconds.min())/2)\n    middle = int(middle_sec*200 + 5000)\n    eeg = pd.read_parquet(parquet_path)\n    t = sec*200\n    eeg = eeg.iloc[middle-t//2:middle+t//2]\n    \n    for k in range(len(FEATS)):\n        COLS = FEATS[k]\n        \n        for kk in range(len(COLS)-1):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            if apply_butter_bandpass_filter:\n                x = butter_bandpass_filter(x, lowcut=0.5, highcut=fmax, fs=200, order=2)\n\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//hop, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=fmax, win_length=win_length)\n\n            width = (mel_spec.shape[1]//32)*32\n            if w_cut:\n                mel_spec = mel_spec[:, :width]\n            if (k==0) & (kk==0):\n                img = np.zeros((mel_spec.shape[0], mel_spec.shape[1], len(FEATS)), dtype='float32')\n\n            img[:,:,k] += mel_spec\n                \n        img[:,:,k] /= (len(COLS)-1)\n        \n    img = np.log(img+1e-9) # log transform\n    np.save(f'{directory_path}/{eeg_id}_{eeg_group_num}.npy',img)\n        \ndef eeg_to_spectrograms(fmax, hop=256, win_length=128, sec=50, apply_butter_bandpass_filter=False):\n    w_cut = True\n    display = False\n    args = [(a, fmax, win_length, hop, w_cut, display, sec, apply_butter_bandpass_filter) for a in list(df.groupby(['eeg_id', 'group_in_eeg']))]\n    print('eeg to spectrograms...')\n    p = multiprocessing.Pool(4)\n    p.map(spectrogram_from_eeg, args)\n    p.close()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:32.166491Z","iopub.execute_input":"2024-04-11T07:26:32.167134Z","iopub.status.idle":"2024-04-11T07:26:32.184189Z","shell.execute_reply.started":"2024-04-11T07:26:32.167097Z","shell.execute_reply":"2024-04-11T07:26:32.182602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATS = [\n['Fp1','F7'],\n['F7','T3'],\n['T3','T5'],\n['T5','O1'],\n['Fp1','F3'],\n['F3','C3'],\n['C3','P3'],\n['P3','O1'],\n['Fp2','F8'],\n['F8','T4'],\n['T4','T6'],\n['T6','O2'],\n['Fp2','F4'],\n['F4','C4'],\n['C4','P4'],\n['P4','O2'],\n]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:32.185639Z","iopub.execute_input":"2024-04-11T07:26:32.186069Z","iopub.status.idle":"2024-04-11T07:26:32.202652Z","shell.execute_reply.started":"2024-04-11T07:26:32.186035Z","shell.execute_reply":"2024-04-11T07:26:32.201627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory_path = f'.'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)\n    \neeg_to_spectrograms(fmax=60, hop=256, win_length=128, sec=40, apply_butter_bandpass_filter=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T07:26:32.203968Z","iopub.execute_input":"2024-04-11T07:26:32.205392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}