{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7537993,"sourceType":"datasetVersion","datasetId":4389508},{"sourceId":7562031,"sourceType":"datasetVersion","datasetId":4403195},{"sourceId":7563702,"sourceType":"datasetVersion","datasetId":4404213},{"sourceId":7572038,"sourceType":"datasetVersion","datasetId":4408129},{"sourceId":7662819,"sourceType":"datasetVersion","datasetId":4412486}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom PIL import Image\nimport gc\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-13T13:20:03.895130Z","iopub.execute_input":"2024-02-13T13:20:03.895552Z","iopub.status.idle":"2024-02-13T13:20:22.231965Z","shell.execute_reply.started":"2024-02-13T13:20:03.895513Z","shell.execute_reply":"2024-02-13T13:20:22.230819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Path definition","metadata":{}},{"cell_type":"code","source":"EEG_TEST_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nSPEC_TEST_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\n\nif not os.path.exists('/kaggle/working/test_eegs_img/'):\n    os.makedirs('/kaggle/working/test_eegs_img/')\nEEG_IMG_TEST_PATH = '/kaggle/working/test_eegs_img/'\n\nif not os.path.exists('/kaggle/working/test_spec_img/'):\n    os.makedirs('/kaggle/working/test_spec_img/')\nSPEC_IMG_TEST_PATH='/kaggle/working/test_spec_img/'\n\nMETA_TEST = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.233833Z","iopub.execute_input":"2024-02-13T13:20:22.234456Z","iopub.status.idle":"2024-02-13T13:20:22.247052Z","shell.execute_reply.started":"2024-02-13T13:20:22.234425Z","shell.execute_reply":"2024-02-13T13:20:22.245840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions to generate images of graphs","metadata":{}},{"cell_type":"code","source":"eeg_zone={\n    'Cz-Pz' : ['Cz', 'Pz'],\n    'Fz-Cz' : ['Fz', 'Cz'],\n    \n    'P4-O2' : ['P4', 'O2'],\n    'C4-P4' : ['C4', 'P4'],\n    'F4-C4' : ['F4', 'C4'],\n    'Fp2-F4': ['Fp2', 'F4'],\n    \n    'P3-O1' : ['P3', 'O1'],\n    'C3-P3' : ['C3', 'P3'],\n    'F3-C3' : ['F3', 'C3'],\n    'Fp1-F3' : ['Fp1', 'F3'],\n    \n    'T6-O2' : ['T6','O2'],\n    'T4-T6' : ['T4', 'T6'],\n    'F8-T4' : ['F8', 'T4'],\n    'Fp2-F8' : ['Fp2', 'F8'],\n    \n    'T5-O1' : ['T5', 'O1'],\n    'T3-T5' : ['T3', 'T5'],\n    'F7-T3' : ['F7', 'T3'],\n    'Fp1-F7' : ['Fp1', 'F7'],\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.255671Z","iopub.execute_input":"2024-02-13T13:20:22.256740Z","iopub.status.idle":"2024-02-13T13:20:22.372789Z","shell.execute_reply.started":"2024-02-13T13:20:22.256688Z","shell.execute_reply":"2024-02-13T13:20:22.371342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_eeg(eegid,input_path=EEG_TEST_PATH,eeg_zone=eeg_zone,linewidth=0.2, eeg_out=EEG_IMG_TEST_PATH):\n    #get eeg data\n    eeg = pd.read_parquet(f'{input_path}{eegid}.parquet')\n    #take subsample of 50sec\n    #eeg = eeg.iloc[int(eegoffset*200):int((eegoffset+50)*200)]\n    \n    # generate the graph\n    ysticks=[]\n    labels= []\n    cicles= 0\n    relpos= 0\n    fig, ax = plt.subplots(1,1, figsize=(3,3), sharex= True)\n    \n    for key, values in eeg_zone.items():\n        ax.plot(eeg.index/200, eeg[values[0]]-eeg[values[1]]+relpos,color='black',linewidth=linewidth)\n        \n        ysticks.append(relpos)\n        labels.append(key)\n        if cicles==1 or cicles==5 or cicles==9 or cicles==13:\n            relpos+=200\n        else:\n            relpos+=40\n            \n        cicles+=1\n       \n    #ax.set_yticks(ysticks, labels=labels)\n    ax.set_xticks([])\n    ax.set_yticks([])\n    \n    ax.set_xlim(0,50)\n    save_path= f'{eeg_out}{eegid}.jpeg' \n    fig.savefig(save_path,bbox_inches='tight', dpi=100)\n    plt.close()\n    \n    return  save_path","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.374242Z","iopub.execute_input":"2024-02-13T13:20:22.374800Z","iopub.status.idle":"2024-02-13T13:20:22.391090Z","shell.execute_reply.started":"2024-02-13T13:20:22.374764Z","shell.execute_reply":"2024-02-13T13:20:22.389854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Spectrogram graph function","metadata":{}},{"cell_type":"code","source":"spec_zones=['LL','RL','LP','RP']\n\ndef generate_spectrogram(specid,\n                         input_path=SPEC_TEST_PATH, spec_out=SPEC_IMG_TEST_PATH,\n                         spec_zones=spec_zones,\n                        output_filter=0.5):\n    \n    #get spec data\n    spec = pd.read_parquet(f'{input_path}{specid}.parquet')\n    spec = spec.fillna(0)\n    #take subsample of 600sec (10min)\n    #spec= spec.loc[(spec.time>=specoffset) & (spec.time<specoffset+600)]\n    #adpat dataset\n    spec=spec.set_index('time')\n    spec=spec.T\n    spec['column']=spec.index.str.split('_', expand=True)\n    \n    spec['freq'] = spec.column.apply(lambda x: x[1]).astype(float)\n    spec['brainreg'] = spec.column.apply(lambda x: x[0]).astype(str)\n    \n    spec=spec.drop('column', axis=1)\n    spec.set_index('freq',inplace=True)\n    #generate subdatases from brain zones\n    subspec=dict()\n    for zone in spec_zones:\n        subspec[f'{zone}_sub']=spec[spec.brainreg==zone]\n        subspec[f'{zone}_sub']= subspec[f'{zone}_sub'].drop('brainreg', axis=1)\n    \n    # generate the graph\n    \n    fig, ax = plt.subplots(nrows=len(spec_zones), figsize=(3,3), sharex=True)\n    for row in range(len(spec_zones)):\n        data=subspec[f'{spec_zones[row]}_sub']\n        ax[row].imshow(data, cmap='turbo', \n                       aspect='auto', \n                       origin='lower', \n                       extent=[data.columns.min(),data.columns.max(),data.index.min(),data.index.max()],\n                      vmin=0,vmax=data.max().max()*output_filter)\n        \n        ax[row].set_xticks([])\n        ax[row].set_yticks([])\n        \n    plt.subplots_adjust(hspace=0.01)\n    \n    save_path= f'{spec_out}{specid}.jpeg'\n    fig.savefig(save_path,bbox_inches='tight', dpi=100)\n    plt.close()\n    #plt.show()\n    return save_path","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.392556Z","iopub.execute_input":"2024-02-13T13:20:22.393480Z","iopub.status.idle":"2024-02-13T13:20:22.408167Z","shell.execute_reply.started":"2024-02-13T13:20:22.393447Z","shell.execute_reply":"2024-02-13T13:20:22.406977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = pd.read_csv(META_TEST)\ntrain_metadata = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.409697Z","iopub.execute_input":"2024-02-13T13:20:22.410040Z","iopub.status.idle":"2024-02-13T13:20:22.738118Z","shell.execute_reply.started":"2024-02-13T13:20:22.410011Z","shell.execute_reply":"2024-02-13T13:20:22.736802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drop_images(paths):\n    for path in paths:\n        os.remove(path)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.739574Z","iopub.execute_input":"2024-02-13T13:20:22.739915Z","iopub.status.idle":"2024-02-13T13:20:22.744738Z","shell.execute_reply.started":"2024-02-13T13:20:22.739886Z","shell.execute_reply":"2024-02-13T13:20:22.743709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_data(metadata,row,train=False):\n    if train:\n        eeg_path = generate_eeg(eegid=metadata.loc[row].eeg_id,input_path='/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/')\n        spec_path = generate_spectrogram(specid=metadata.loc[row].spectrogram_id,output_filter=0.7,input_path='/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/')\n        # create image_tensor\n    else:\n        eeg_path=generate_eeg(eegid=metadata.loc[row].eeg_id)\n        spec_path=generate_spectrogram(specid=metadata.loc[row].spectrogram_id,output_filter=0.7)\n    \n    \n    eeg_img = Image.open(eeg_path)\n    spec_img = Image.open(spec_path)\n    \n    eeg_img_tr = np.array(eeg_img)\n    spec_img_tr = np.array(spec_img)\n        \n    #eeg_img_tr = tf.image.per_image_standardization(eeg_img)\n    #spec_img_tr = tf.image.per_image_standardization(spec_img)\n    \n        \n    eeg_img_tr =tf.convert_to_tensor(eeg_img)\n    spec_img_tr =tf.convert_to_tensor(spec_img)\n    \n    eeg_img_tr =tf.image.resize(eeg_img_tr, [224,224])\n    spec_img_tr = tf.image.resize(spec_img_tr, [240,240])\n    \n    eeg_img_tr = tf.expand_dims(eeg_img_tr, axis=0)\n    spec_img_tr = tf.expand_dims(spec_img_tr, axis=0)\n    \n    drop_images(paths=[eeg_path,spec_path])\n    eeg_id = metadata.loc[row].eeg_id\n    del metadata\n    #images= [eeg_img_tr, spec_img_tr]\n    return eeg_img_tr,spec_img_tr, eeg_id","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:26:12.923872Z","iopub.execute_input":"2024-02-13T13:26:12.924314Z","iopub.status.idle":"2024-02-13T13:26:12.936570Z","shell.execute_reply.started":"2024-02-13T13:26:12.924268Z","shell.execute_reply":"2024-02-13T13:26:12.935197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model(\"/kaggle/input/hms-models/best_model_aug_01.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:20:22.763548Z","iopub.execute_input":"2024-02-13T13:20:22.763974Z","iopub.status.idle":"2024-02-13T13:22:24.120139Z","shell.execute_reply.started":"2024-02-13T13:20:22.763946Z","shell.execute_reply":"2024-02-13T13:22:24.118865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission={\n    'eeg_id' : [],\n    'seizure_vote' : [],\n    'lpd_vote' : [],\n    'gpd_vote' : [],\n    'lrda_vote' : [],\n    'grda_vote' : [],\n    'other_vote' : [],\n}\niter_dict = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']\n\n%matplotlib agg\nfor row in metadata.index:\n    \n    eeg_img, spec_img, eeg_id = preprocess_data(metadata,row)\n    prediction = model.predict([eeg_img,spec_img],verbose=1)\n    prediction = prediction.squeeze()\n    submission['eeg_id'].append(eeg_id)\n    counter=0\n    for illness in iter_dict:\n        submission[illness].append(prediction[counter])\n        counter+=1\n        \npdsubmit=pd.DataFrame(submission)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:26:15.925156Z","iopub.execute_input":"2024-02-13T13:26:15.925824Z","iopub.status.idle":"2024-02-13T13:26:21.450359Z","shell.execute_reply.started":"2024-02-13T13:26:15.925784Z","shell.execute_reply":"2024-02-13T13:26:21.449377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdsubmit","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:26:25.123944Z","iopub.execute_input":"2024-02-13T13:26:25.124388Z","iopub.status.idle":"2024-02-13T13:26:25.144520Z","shell.execute_reply.started":"2024-02-13T13:26:25.124352Z","shell.execute_reply":"2024-02-13T13:26:25.143655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdsubmit.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T13:22:25.981168Z","iopub.status.idle":"2024-02-13T13:22:25.981974Z","shell.execute_reply.started":"2024-02-13T13:22:25.981674Z","shell.execute_reply":"2024-02-13T13:22:25.981701Z"},"trusted":true},"execution_count":null,"outputs":[]}]}