{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-06T13:59:03.699336Z","iopub.execute_input":"2024-03-06T13:59:03.700739Z","iopub.status.idle":"2024-03-06T13:59:15.875878Z","shell.execute_reply.started":"2024-03-06T13:59:03.700687Z","shell.execute_reply":"2024-03-06T13:59:15.874398Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf.head()\ndf.shape\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:51:25.990873Z","iopub.execute_input":"2024-03-07T14:51:25.992589Z","iopub.status.idle":"2024-03-07T14:51:26.244874Z","shell.execute_reply.started":"2024-03-07T14:51:25.992532Z","shell.execute_reply":"2024-03-07T14:51:26.242806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ydata_profiling import ProfileReport\nprofile = ProfileReport(df)\nprofile","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:37:14.363888Z","iopub.execute_input":"2024-03-07T14:37:14.365300Z","iopub.status.idle":"2024-03-07T14:38:00.133998Z","shell.execute_reply.started":"2024-03-07T14:37:14.365245Z","shell.execute_reply":"2024-03-07T14:38:00.132013Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Non-null count have the same value \nhence\n+ No missing value","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nexample_parquet = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000189855.parquet\")\nprint(example_parquet)","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:52:51.469585Z","iopub.execute_input":"2024-03-07T14:52:51.470093Z","iopub.status.idle":"2024-03-07T14:52:51.567014Z","shell.execute_reply.started":"2024-03-07T14:52:51.470057Z","shell.execute_reply":"2024-03-07T14:52:51.565771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom PIL import Image\n#example_parquet = example_parquet.to_numpy()\nimage = Image.fromarray(example_parquet, mode=\"L\" if len(example_parquet.shape) == 3 else \"L\")\nplt.imshow(image)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-07T15:13:20.141386Z","iopub.execute_input":"2024-03-07T15:13:20.142628Z","iopub.status.idle":"2024-03-07T15:13:20.429885Z","shell.execute_reply.started":"2024-03-07T15:13:20.142593Z","shell.execute_reply":"2024-03-07T15:13:20.428350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_parquet.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-07T15:13:38.290368Z","iopub.execute_input":"2024-03-07T15:13:38.290790Z","iopub.status.idle":"2024-03-07T15:13:38.299405Z","shell.execute_reply.started":"2024-03-07T15:13:38.290758Z","shell.execute_reply":"2024-03-07T15:13:38.298155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = len(os.listdir('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms'))\nb = len(os.listdir('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'))\nprint(a,b)","metadata":{"execution":{"iopub.status.busy":"2024-03-07T15:00:35.101338Z","iopub.execute_input":"2024-03-07T15:00:35.101764Z","iopub.status.idle":"2024-03-07T15:00:35.122996Z","shell.execute_reply.started":"2024-03-07T15:00:35.101733Z","shell.execute_reply":"2024-03-07T15:00:35.121808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a= {5:1,3:2}\nprint(a.values)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T14:22:12.365764Z","iopub.execute_input":"2024-03-06T14:22:12.366237Z","iopub.status.idle":"2024-03-06T14:22:12.373146Z","shell.execute_reply.started":"2024-03-06T14:22:12.366203Z","shell.execute_reply":"2024-03-06T14:22:12.371750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a[0]=2\nprint(a)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T14:17:29.034564Z","iopub.execute_input":"2024-03-06T14:17:29.035003Z","iopub.status.idle":"2024-03-06T14:17:29.042681Z","shell.execute_reply.started":"2024-03-06T14:17:29.034973Z","shell.execute_reply":"2024-03-06T14:17:29.041534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:38:11.844564Z","iopub.execute_input":"2024-03-07T14:38:11.844983Z","iopub.status.idle":"2024-03-07T14:38:11.851069Z","shell.execute_reply.started":"2024-03-07T14:38:11.844952Z","shell.execute_reply":"2024-03-07T14:38:11.849547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\ndirectory_path = 'EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)\ndef spectrogram_from_eeg(parquet_path, USE_WAVELET=False, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            #if USE_WAVELET:\n             #   x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:42:55.799430Z","iopub.execute_input":"2024-03-07T14:42:55.799891Z","iopub.status.idle":"2024-03-07T14:42:55.822508Z","shell.execute_reply.started":"2024-03-07T14:42:55.799860Z","shell.execute_reply":"2024-03-07T14:42:55.820949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spectrogram_from_eeg(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet\", display=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:43:55.918869Z","iopub.execute_input":"2024-03-07T14:43:55.919270Z","iopub.status.idle":"2024-03-07T14:43:57.974853Z","shell.execute_reply.started":"2024-03-07T14:43:55.919232Z","shell.execute_reply":"2024-03-07T14:43:57.973317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_IDS= [\"1000913311\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:40:36.750911Z","iopub.execute_input":"2024-03-07T14:40:36.751310Z","iopub.status.idle":"2024-03-07T14:40:36.757447Z","shell.execute_reply.started":"2024-03-07T14:40:36.751282Z","shell.execute_reply":"2024-03-07T14:40:36.755791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape', train.shape )\ndisplay( train.head() )\n\n#%%time\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\n#EEG_IDS = train.eeg_id.unique()\nall_eegs = {}\n#PRINT PROCESS\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='')\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET, i<DISPLAY:  waiting for 4 pics created\n    img = spectrogram_from_eeg(f'{PATH}{eeg_id}.parquet', i<DISPLAY)\n    \n    # SAVE TO DISK\n    if i==DISPLAY:\n        print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n    np.save(f'{directory_path}{eeg_id}',img)\n    all_eegs[eeg_id] = img\n   \n# SAVE EEG SPECTROGRAM DICTIONARY\nnp.save('eeg_specs',all_eegs)","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:43:07.549214Z","iopub.execute_input":"2024-03-07T14:43:07.549668Z","iopub.status.idle":"2024-03-07T14:43:08.101441Z","shell.execute_reply.started":"2024-03-07T14:43:07.549637Z","shell.execute_reply":"2024-03-07T14:43:08.099175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_IDS = train.eeg_id.unique()\nprint(EEG_IDS)","metadata":{"execution":{"iopub.status.busy":"2024-03-07T14:43:13.956196Z","iopub.execute_input":"2024-03-07T14:43:13.957572Z","iopub.status.idle":"2024-03-07T14:43:13.970555Z","shell.execute_reply.started":"2024-03-07T14:43:13.957506Z","shell.execute_reply":"2024-03-07T14:43:13.969199Z"},"trusted":true},"execution_count":null,"outputs":[]}]}