{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7483964,"sourceType":"datasetVersion","datasetId":4356750},{"sourceId":185,"sourceType":"modelInstanceVersion","modelInstanceId":132}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-15T01:39:44.595470Z","iopub.execute_input":"2024-02-15T01:39:44.595755Z","iopub.status.idle":"2024-02-15T01:39:45.760912Z","shell.execute_reply.started":"2024-02-15T01:39:44.595728Z","shell.execute_reply":"2024-02-15T01:39:45.759981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Face into data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:45.762726Z","iopub.execute_input":"2024-02-15T01:39:45.763104Z","iopub.status.idle":"2024-02-15T01:39:46.994444Z","shell.execute_reply.started":"2024-02-15T01:39:45.763078Z","shell.execute_reply":"2024-02-15T01:39:46.993461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint(f'Train data shape = {train.shape}')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:46.995720Z","iopub.execute_input":"2024-02-15T01:39:46.996055Z","iopub.status.idle":"2024-02-15T01:39:47.356554Z","shell.execute_reply.started":"2024-02-15T01:39:46.996023Z","shell.execute_reply":"2024-02-15T01:39:47.355542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Count of unique patients: {len(train.patient_id.unique())}')\nprint(f'Count of unique spectrograms: {len(train.spectrogram_id.unique())}')\nprint(f'Count of unique EEGs: {len(train.eeg_id.unique())}')","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.359725Z","iopub.execute_input":"2024-02-15T01:39:47.360084Z","iopub.status.idle":"2024-02-15T01:39:47.374055Z","shell.execute_reply.started":"2024-02-15T01:39:47.360053Z","shell.execute_reply":"2024-02-15T01:39:47.372965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_columns = ['eeg_id','eeg_sub_id','eeg_label_offset_seconds','spectrogram_id','spectrogram_sub_id','spectrogram_label_offset_seconds','label_id','patient_id']","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.375357Z","iopub.execute_input":"2024-02-15T01:39:47.375883Z","iopub.status.idle":"2024-02-15T01:39:47.381582Z","shell.execute_reply.started":"2024-02-15T01:39:47.375850Z","shell.execute_reply":"2024-02-15T01:39:47.380319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_COLUMNS = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']\nCLASS_NAMES = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\nLABEL2NAME = dict(enumerate(CLASS_NAMES))\nNAME2LABEL = {v:k for k, v in LABEL2NAME.items()}\n\nEEG_PATH_TEMPL = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nSP_PATH_TEMPL = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n\nWIN_SIZE =  10 # 10 seconds\nEEG_FR = 200 # 200 samples per seconds\nEEG_T = WIN_SIZE*EEG_FR\nCHAINS = {\n    'LL' : [(\"Fp1\",\"F7\"),(\"F7\",\"T3\"),(\"T3\",\"T5\"),(\"T5\",\"O1\")],\n    'RL' : [(\"Fp2\",\"F8\"),(\"F8\",\"T4\"),(\"T4\",\"T6\"),(\"T6\",\"O2\")],\n    'LP' : [(\"Fp1\",\"F3\"),(\"F3\",\"C3\"),(\"C3\",\"P3\"),(\"P3\",\"O1\")],\n    'RP' : [(\"Fp2\",\"F4\"),(\"F4\",\"C4\"),(\"C4\",\"P4\"),(\"P4\",\"O2\")]\n}\nSP_WIN = 600 # 10 minutes = 600 seconds\nEGG_WIN = 50 # 50 seconds\n\nLABELED_SECS = 10\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.382983Z","iopub.execute_input":"2024-02-15T01:39:47.383349Z","iopub.status.idle":"2024-02-15T01:39:47.396786Z","shell.execute_reply.started":"2024-02-15T01:39:47.383293Z","shell.execute_reply":"2024-02-15T01:39:47.395595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization data","metadata":{}},{"cell_type":"code","source":"def get_eeg_sp_data(train_row):\n    eeg_id = train_row.eeg_id\n    sp_id = train_row.spectrogram_id\n    \n    eeg_parquet = pd.read_parquet(f'{EEG_PATH_TEMPL}{eeg_id}.parquet')\n    sp_parquet = pd.read_parquet(f'{SP_PATH_TEMPL}{sp_id}.parquet')\n    \n    # offset of data\n    eeg_offset = int(train_row.eeg_label_offset_seconds + 20) #only 10 central seconds from 50 secs were labeled\n    sp_offset = int(train_row.spectrogram_label_offset_seconds )\n    \n    # get spectrogram data\n    sp = sp_parquet.loc[(sp_parquet.time>=sp_offset)&(sp_parquet.time<sp_offset+SP_WIN)]\n    sp = sp.loc[:, sp.columns != 'time']\n    sp = {\n        \"LL\": sp.filter(regex='^LL', axis=1),\n        \"RL\": sp.filter(regex='^RL', axis=1),\n        \"RP\": sp.filter(regex='^RP', axis=1),\n        \"LP\": sp.filter(regex='^LP', axis=1)}\n    \n    # calculate eeg data\n    eeg_data = eeg_parquet.iloc[eeg_offset*EEG_FR:(eeg_offset+WIN_SIZE)*EEG_FR]\n    \n    eeg = {}\n    for chain in CHAINS.keys():\n        eeg[chain] = []\n        for s_i, signals in enumerate(CHAINS[chain]):\n            diff=eeg_data[signals[0]]-eeg_data[signals[1]]\n            diff.ffill(inplace = True)\n            eeg[chain].append(diff)\n    \n    return eeg, sp, train_row[TARGET_COLUMNS].values","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.398052Z","iopub.execute_input":"2024-02-15T01:39:47.398845Z","iopub.status.idle":"2024-02-15T01:39:47.411575Z","shell.execute_reply.started":"2024-02-15T01:39:47.398810Z","shell.execute_reply":"2024-02-15T01:39:47.410576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_id = 4567\nexp_row = train.iloc[example_id]\nexp_row","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.413180Z","iopub.execute_input":"2024-02-15T01:39:47.413574Z","iopub.status.idle":"2024-02-15T01:39:47.430705Z","shell.execute_reply.started":"2024-02-15T01:39:47.413538Z","shell.execute_reply":"2024-02-15T01:39:47.429699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_data, sp_data, targets = get_eeg_sp_data(exp_row)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.431927Z","iopub.execute_input":"2024-02-15T01:39:47.432818Z","iopub.status.idle":"2024-02-15T01:39:47.777993Z","shell.execute_reply.started":"2024-02-15T01:39:47.432792Z","shell.execute_reply":"2024-02-15T01:39:47.776767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_data['LL'][0].shape, sp_data['LL'].shape, targets","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.781026Z","iopub.execute_input":"2024-02-15T01:39:47.781346Z","iopub.status.idle":"2024-02-15T01:39:47.788093Z","shell.execute_reply.started":"2024-02-15T01:39:47.781295Z","shell.execute_reply":"2024-02-15T01:39:47.787122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.gridspec import GridSpec\n \n# create objects\n\ndef plot_data(eeg_data, sp_data):\n    fig, axes = plt.subplots(ncols=2, nrows=len(CHAINS)*4,figsize=(30, 40))\n    \n    time_x = np.arange(-5,5,1/200)\n    x_ticks = np.arange(-5,5,1)\n    \n    for i, chain in enumerate(CHAINS):\n        # plot eeg raw signals\n        for j, dt in enumerate(eeg_data[chain]):\n            ax = sns.lineplot(x=time_x, y=dt, ax=axes[i*4+j, 0])\n            ax.set_xticks(x_ticks)\n            ax.set_title(f'{CHAINS[chain][i][0]}-{CHAINS[chain][i][1]}')\n            ax.grid(True) \n        \n        # plot spectrogram\n        gs = axes[i*4, 1].get_gridspec()\n        axsbig = fig.add_subplot(gs[i*4:(i+1)*4, -1])\n        log_spec = np.log(sp_data[chain].T + np.finfo(float).eps)\n        height = log_spec.shape[0]\n        width = log_spec.shape[1]\n        X = np.linspace(0, np.size(sp_data[chain]), num=width, dtype=int)\n        Y = range(height)\n        axsbig.pcolormesh(X, Y, log_spec)\n        axsbig.set_title(chain)\n    fig.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.789396Z","iopub.execute_input":"2024-02-15T01:39:47.790491Z","iopub.status.idle":"2024-02-15T01:39:47.804285Z","shell.execute_reply.started":"2024-02-15T01:39:47.790454Z","shell.execute_reply":"2024-02-15T01:39:47.803434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'TARGET = {targets}')\nplot_data(eeg_data, sp_data)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:47.805604Z","iopub.execute_input":"2024-02-15T01:39:47.806275Z","iopub.status.idle":"2024-02-15T01:39:56.320187Z","shell.execute_reply.started":"2024-02-15T01:39:47.806240Z","shell.execute_reply":"2024-02-15T01:39:56.319090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_spectrogram(spectrogram, ax):\n    assert len(spectrogram.shape) == 2   \n    # Convert the frequencies to log scale and transpose, so that the time is\n    # represented on the x-axis (columns).\n    # Add an epsilon to avoid taking a log of zero.\n    log_spec = np.log(spectrogram.T + np.finfo(float).eps)\n    height = log_spec.shape[0]\n    width = log_spec.shape[1]\n    X = np.linspace(0, np.size(spectrogram), num=width, dtype=int)\n    Y = range(height)\n    ax.pcolormesh(X, Y, log_spec)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:56.321470Z","iopub.execute_input":"2024-02-15T01:39:56.321775Z","iopub.status.idle":"2024-02-15T01:39:56.327939Z","shell.execute_reply.started":"2024-02-15T01:39:56.321746Z","shell.execute_reply":"2024-02-15T01:39:56.327059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, figsize=(20, 35))\nfor i, chain in enumerate(CHAINS.keys()):\n    plot_spectrogram(sp_data[chain], axes[i])\n    axes[i].set_title(chain)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:56.329126Z","iopub.execute_input":"2024-02-15T01:39:56.329436Z","iopub.status.idle":"2024-02-15T01:39:57.515795Z","shell.execute_reply.started":"2024-02-15T01:39:56.329410Z","shell.execute_reply":"2024-02-15T01:39:57.514800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model1 \nIn this model the spectrogram data is used","metadata":{}},{"cell_type":"markdown","source":"## Read all spectrograms","metadata":{}},{"cell_type":"code","source":"print(f'Shape of a spectrogram is {sp_data[\"LL\"].shape}')","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:57.517155Z","iopub.execute_input":"2024-02-15T01:39:57.517560Z","iopub.status.idle":"2024-02-15T01:39:57.522766Z","shell.execute_reply.started":"2024-02-15T01:39:57.517525Z","shell.execute_reply":"2024-02-15T01:39:57.521797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Read all spectrograms\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nfiles = os.listdir(SP_PATH_TEMPL)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in tqdm(enumerate(files)):\n        tmp = pd.read_parquet(f'{SP_PATH_TEMPL}{f}')\n        sp_id = int(f.split('.')[0])\n        spectrograms[sp_id] = tmp.iloc[:,1:].values\n        #with open(\"/kaggle/working/specs.npy\", \"wb\") as f:\n            #np.save(f, spectrograms)\nelse:\n    spectrograms = np.load('/kaggle/input/all-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:39:57.524060Z","iopub.execute_input":"2024-02-15T01:39:57.524403Z","iopub.status.idle":"2024-02-15T01:41:13.102483Z","shell.execute_reply.started":"2024-02-15T01:39:57.524376Z","shell.execute_reply":"2024-02-15T01:41:13.101520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(spectrograms)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:41:13.103842Z","iopub.execute_input":"2024-02-15T01:41:13.104654Z","iopub.status.idle":"2024-02-15T01:41:13.111029Z","shell.execute_reply.started":"2024-02-15T01:41:13.104612Z","shell.execute_reply":"2024-02-15T01:41:13.110096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a DataReader","metadata":{}},{"cell_type":"code","source":"SPECTROGRAM_SHAPE = (300,400)\nOUTPUT_SHAPE = 6","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:42:21.286660Z","iopub.execute_input":"2024-02-15T01:42:21.287014Z","iopub.status.idle":"2024-02-15T01:42:21.291907Z","shell.execute_reply.started":"2024-02-15T01:42:21.286986Z","shell.execute_reply":"2024-02-15T01:42:21.290887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert target value to propabilities\ny_data = train[TARGET_COLUMNS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGET_COLUMNS] = y_data\ntrain[1000: 1010]","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:42:24.971955Z","iopub.execute_input":"2024-02-15T01:42:24.972290Z","iopub.status.idle":"2024-02-15T01:42:25.006767Z","shell.execute_reply.started":"2024-02-15T01:42:24.972264Z","shell.execute_reply":"2024-02-15T01:42:25.005868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator:\n    'Generates data for Keras'\n    def __init__(self, eeg_data, specs, mode='train', specs_shape = SPECTROGRAM_SHAPE, output_shape = OUTPUT_SHAPE): \n\n        self.eeg_data = eeg_data\n        self.mode = mode\n        self.specs = specs\n        self.specs_shape = specs_shape\n        self.height = self.specs_shape[0]\n        self.width = self.specs_shape[1]\n        self.output_shape = output_shape\n        self.indexes = np.arange( len(self.eeg_data) )\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return len(self.eeg_data)\n    \n    def __call__(self):\n        for j,i in enumerate(self.indexes):\n            yield self.__getitem__(i)\n\n            if j == self.__len__()-1:\n                self.on_epoch_end()\n\n    def __getitem__(self, index):\n        return self.__data_generation(index)\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, index):\n        'Generates data containing batch_size samples' \n        \n        X = np.zeros((self.height,self.width,1),dtype='float32')\n        y = np.zeros((self.output_shape,),dtype='float32')\n        \n        row = self.eeg_data.iloc[index]\n        \n        # offset of data\n        sp_offset = 0\n        if self.mode == 'train':\n            sp_offset = int(row.spectrogram_label_offset_seconds )//2 \n\n        # get spectrogram data\n            # EXTRACT 300 ROWS OF SPECTROGRAM\n        img = self.specs[row.spectrogram_id][sp_offset:sp_offset+self.height,0: self.width]\n        X[:,:,0] = np.nan_to_num(img, nan=0.0)\n\n        if self.mode!='test':\n            y = row[TARGET_COLUMNS]\n            \n        return X,y","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:42:35.761913Z","iopub.execute_input":"2024-02-15T01:42:35.762276Z","iopub.status.idle":"2024-02-15T01:42:35.776376Z","shell.execute_reply.started":"2024-02-15T01:42:35.762245Z","shell.execute_reply":"2024-02-15T01:42:35.775334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nclass DataGenerator:\n    'Generates data for Keras'\n    def __init__(self, eeg_data, specs, mode='train', specs_shape = SPECTROGRAM_SHAPE, output_shape = OUTPUT_SHAPE): \n\n        self.eeg_data = eeg_data\n        self.mode = mode\n        self.specs = specs\n        self.specs_shape = specs_shape\n        self.height = self.specs_shape[0]\n        self.width = self.specs_shape[1]\n        self.output_shape = output_shape\n        self.indexes = np.arange( len(self.eeg_data) )\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return len(self.eeg_data)\n    \n    def __call__(self):\n        for j,i in enumerate(self.indexes):\n            yield self.__getitem__(i)\n\n            if j == self.__len__()-1:\n                self.on_epoch_end()\n\n    def __getitem__(self, index):\n        return self.__data_generation(index)\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, index):\n        'Generates data containing batch_size samples' \n        \n        X = tf.zeros((self.height,self.width,4),dtype='float32')\n        y = tf.zeros((self.output_shape,),dtype='float32')\n        img = np.ones(self.specs_shape,dtype='float32')\n        \n        row = self.eeg_data.iloc[index]\n        \n        # offset of data\n        sp_offset = 0\n        if self.mode == 'train':\n            sp_offset = int(row.spectrogram_label_offset_seconds )//2 \n\n        # get spectrogram data\n        for k in range(4):\n            # EXTRACT 300 ROWS OF SPECTROGRAM\n            img = self.specs[row.spectrogram_id][sp_offset:sp_offset+self.height,k*self.width:(k+1)*self.width]\n\n            # NORMALIZATION PER IMAGE\n            ep = 1e-6\n            m = np.nanmean(img)\n            s = np.nanstd(img)\n            img = (img-m)/(s+ep)\n            #img = tf.image.per_image_standardization(img)\n            img = np.nan_to_num(img, nan=0.0)\n\n            X[:,:,k] = img\n\n                \n            #X[j] = tf.image.per_image_standardization(X[j,:,:,:])\n            if self.mode!='test':\n                y = row[TARGET_COLUMNS]\n            \n        return X,y\n        '''","metadata":{"jupyter":{"outputs_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T01:12:46.141531Z","iopub.execute_input":"2024-02-03T01:12:46.141800Z","iopub.status.idle":"2024-02-03T01:12:46.155992Z","shell.execute_reply.started":"2024-02-03T01:12:46.141777Z","shell.execute_reply":"2024-02-03T01:12:46.155173Z"},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:42:41.074450Z","iopub.execute_input":"2024-02-15T01:42:41.075703Z","iopub.status.idle":"2024-02-15T01:42:54.389588Z","shell.execute_reply.started":"2024-02-15T01:42:41.075655Z","shell.execute_reply":"2024-02-15T01:42:54.388606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_gen = DataGenerator(train, spectrograms)\ntf_ds =tf.data.Dataset.from_generator(data_gen, \n                                      output_types = (tf.float32, tf.float32)).shuffle(1000).batch(128).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:42:56.883577Z","iopub.execute_input":"2024-02-15T01:42:56.884750Z","iopub.status.idle":"2024-02-15T01:42:58.074614Z","shell.execute_reply.started":"2024-02-15T01:42:56.884712Z","shell.execute_reply":"2024-02-15T01:42:58.073789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor (x,y) in tf_ds:\n    print(x.shape)\n    print(y.shape)\n    break","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:43:31.292081Z","iopub.execute_input":"2024-02-15T01:43:31.292509Z","iopub.status.idle":"2024-02-15T01:43:33.461130Z","shell.execute_reply.started":"2024-02-15T01:43:31.292468Z","shell.execute_reply":"2024-02-15T01:43:33.460086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimg1 = x[5]\nlabel1 = y[5]\n\nfig, ax = plt.subplots(1, figsize=(22, 20))\nprint(label1)\nplot_spectrogram(img1[:,:,0].numpy(), ax)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:43:36.457787Z","iopub.execute_input":"2024-02-15T01:43:36.458132Z","iopub.status.idle":"2024-02-15T01:43:37.100198Z","shell.execute_reply.started":"2024-02-15T01:43:36.458104Z","shell.execute_reply":"2024-02-15T01:43:37.099261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create train and validation datasets","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.layers import Input, Dense, Multiply, Add, Conv2D, AveragePooling2D,Normalization,MaxPooling2D,Dropout,Flatten\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:43:44.892326Z","iopub.execute_input":"2024-02-15T01:43:44.892692Z","iopub.status.idle":"2024-02-15T01:43:45.169691Z","shell.execute_reply.started":"2024-02-15T01:43:44.892660Z","shell.execute_reply":"2024-02-15T01:43:45.168669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, val_data = train_test_split(train, test_size=0.30, random_state=1234)\ntrain_data.shape, val_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:43:47.836823Z","iopub.execute_input":"2024-02-15T01:43:47.837233Z","iopub.status.idle":"2024-02-15T01:43:47.867618Z","shell.execute_reply.started":"2024-02-15T01:43:47.837197Z","shell.execute_reply":"2024-02-15T01:43:47.866602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_img(img, label, img_shape=(128,128)):\n    # normalization\n    img = tf.image.resize(img, [img_shape[0], img_shape[1]])\n    img = tf.image.per_image_standardization(img)\n    return img, label","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:32.675522Z","iopub.execute_input":"2024-02-15T01:45:32.676248Z","iopub.status.idle":"2024-02-15T01:45:32.685643Z","shell.execute_reply.started":"2024-02-15T01:45:32.676196Z","shell.execute_reply":"2024-02-15T01:45:32.684745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds_gen = DataGenerator(train_data, spectrograms)\nval_ds_gen = DataGenerator(val_data, spectrograms)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:35.099412Z","iopub.execute_input":"2024-02-15T01:45:35.100240Z","iopub.status.idle":"2024-02-15T01:45:35.108909Z","shell.execute_reply.started":"2024-02-15T01:45:35.100194Z","shell.execute_reply":"2024-02-15T01:45:35.107631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = tf.data.Dataset.from_generator(train_ds_gen,output_types = (tf.float32, tf.float32), output_shapes =([300 , 400 , 1] , [6 , ]))\ntrain_ds = train_ds.map(preprocess_img)\\\n        .shuffle(1000) \\\n        .batch(32) \\\n        .prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:37.202772Z","iopub.execute_input":"2024-02-15T01:45:37.203128Z","iopub.status.idle":"2024-02-15T01:45:37.330022Z","shell.execute_reply.started":"2024-02-15T01:45:37.203098Z","shell.execute_reply":"2024-02-15T01:45:37.329003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_ds = tf.data.Dataset.from_generator(val_ds_gen,output_types = (tf.float32, tf.float32), output_shapes =([300 , 400 , 1] , [6 , ]))\nval_ds = val_ds.map(preprocess_img)\\\n        .shuffle(1000) \\\n        .batch(32) \\\n        .prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:39.391264Z","iopub.execute_input":"2024-02-15T01:45:39.392282Z","iopub.status.idle":"2024-02-15T01:45:39.453620Z","shell.execute_reply.started":"2024-02-15T01:45:39.392243Z","shell.execute_reply":"2024-02-15T01:45:39.452361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lenet_model(input_shape,num_labels):\n    model = tf.keras.Sequential()\n    model.add(Conv2D(filters=16, kernel_size=(3, 3), activation='relu', input_shape=(128,128,1)))\n    model.add(AveragePooling2D())\n    model.add(Conv2D(filters=32, kernel_size=(3, 3), activation='relu'))\n    model.add(AveragePooling2D())\n    model.add(Flatten())\n    model.add(Dense(units=120, activation='relu'))\n    model.add(Dropout(0.5))\n    model.add(Dense(units=84, activation='relu'))\n    model.add(Dropout(0.5))\n    model.add(Dense(units=num_labels, activation='softmax', dtype='float32'))\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:44.331274Z","iopub.execute_input":"2024-02-15T01:45:44.331657Z","iopub.status.idle":"2024-02-15T01:45:44.339468Z","shell.execute_reply.started":"2024-02-15T01:45:44.331628Z","shell.execute_reply":"2024-02-15T01:45:44.338462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE MULTIPLE GPUS\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:47.418526Z","iopub.execute_input":"2024-02-15T01:45:47.419226Z","iopub.status.idle":"2024-02-15T01:45:47.715966Z","shell.execute_reply.started":"2024-02-15T01:45:47.419187Z","shell.execute_reply":"2024-02-15T01:45:47.714956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = get_lenet_model(SPECTROGRAM_SHAPE,OUTPUT_SHAPE)\n    opt = tf.keras.optimizers.Adam(learning_rate = 1e-3)\n    loss = tf.keras.losses.KLDivergence()\n\n    model.compile(loss=loss, optimizer = opt)\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:45:55.146472Z","iopub.execute_input":"2024-02-15T01:45:55.147031Z","iopub.status.idle":"2024-02-15T01:45:55.428924Z","shell.execute_reply.started":"2024-02-15T01:45:55.146988Z","shell.execute_reply":"2024-02-15T01:45:55.427642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, verbose=1,\n    validation_data = val_ds,\n    steps_per_epoch = 250,\n    validation_steps = 125, epochs=120)\n#model.save('/kaggle/working/model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:46:11.627154Z","iopub.execute_input":"2024-02-15T01:46:11.628066Z","iopub.status.idle":"2024-02-15T01:49:41.996398Z","shell.execute_reply.started":"2024-02-15T01:46:11.628028Z","shell.execute_reply":"2024-02-15T01:49:41.995458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(val_ds)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T01:49:54.600575Z","iopub.execute_input":"2024-02-15T01:49:54.600951Z","iopub.status.idle":"2024-02-15T01:50:45.846471Z","shell.execute_reply.started":"2024-02-15T01:49:54.600919Z","shell.execute_reply":"2024-02-15T01:50:45.845548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.save('/kaggle/working/model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-02-03T02:54:06.201948Z","iopub.execute_input":"2024-02-03T02:54:06.202331Z","iopub.status.idle":"2024-02-03T02:54:06.308287Z","shell.execute_reply.started":"2024-02-03T02:54:06.202299Z","shell.execute_reply":"2024-02-03T02:54:06.307362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read test_data\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint(f'Test data shape = {test.shape}')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T02:57:09.968723Z","iopub.execute_input":"2024-02-03T02:57:09.969416Z","iopub.status.idle":"2024-02-03T02:57:09.985744Z","shell.execute_reply.started":"2024-02-03T02:57:09.969378Z","shell.execute_reply":"2024-02-03T02:57:09.984841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sp_id = test.iloc[0].spectrogram_id\ntest_sp_id","metadata":{"execution":{"iopub.status.busy":"2024-02-03T02:57:12.355679Z","iopub.execute_input":"2024-02-03T02:57:12.356068Z","iopub.status.idle":"2024-02-03T02:57:12.362930Z","shell.execute_reply.started":"2024-02-03T02:57:12.356033Z","shell.execute_reply":"2024-02-03T02:57:12.361914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sp = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/{test_sp_id}.parquet')\ntest_spectrogram = {test_sp_id: test_sp.iloc[:,1:].values}","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:00:38.953579Z","iopub.execute_input":"2024-02-03T03:00:38.953962Z","iopub.status.idle":"2024-02-03T03:00:38.990446Z","shell.execute_reply.started":"2024-02-03T03:00:38.953930Z","shell.execute_reply":"2024-02-03T03:00:38.989721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_spectrogram[test_sp_id].shape","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:01:07.188760Z","iopub.execute_input":"2024-02-03T03:01:07.189522Z","iopub.status.idle":"2024-02-03T03:01:07.195517Z","shell.execute_reply.started":"2024-02-03T03:01:07.189489Z","shell.execute_reply":"2024-02-03T03:01:07.194618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sp = tf.convert_to_tensor(test_spectrogram[test_sp_id])\ntest_sp = tf.expand_dims(test_sp, axis = 2)\ntest_sp = preprocess_img(test_sp,None)\ntets_data = test_sp[0].numpy()\ntets_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:33:53.725602Z","iopub.execute_input":"2024-02-03T03:33:53.726461Z","iopub.status.idle":"2024-02-03T03:33:53.738389Z","shell.execute_reply.started":"2024-02-03T03:33:53.726427Z","shell.execute_reply":"2024-02-03T03:33:53.737628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(tets_data)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:33:57.049434Z","iopub.execute_input":"2024-02-03T03:33:57.050288Z","iopub.status.idle":"2024-02-03T03:33:57.339196Z","shell.execute_reply.started":"2024-02-03T03:33:57.050255Z","shell.execute_reply":"2024-02-03T03:33:57.338241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.std(tets_data[:,:,0])","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:25:44.491666Z","iopub.execute_input":"2024-02-03T03:25:44.492272Z","iopub.status.idle":"2024-02-03T03:25:44.500729Z","shell.execute_reply.started":"2024-02-03T03:25:44.492227Z","shell.execute_reply":"2024-02-03T03:25:44.499621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tets_data = np.expand_dims(tets_data,0)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:35:18.469179Z","iopub.execute_input":"2024-02-03T03:35:18.469533Z","iopub.status.idle":"2024-02-03T03:35:18.473997Z","shell.execute_reply.started":"2024-02-03T03:35:18.469508Z","shell.execute_reply":"2024-02-03T03:35:18.472971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction\nprediction = model.predict(tets_data)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:16.745394Z","iopub.execute_input":"2024-02-03T03:36:16.745751Z","iopub.status.idle":"2024-02-03T03:36:16.932878Z","shell.execute_reply.started":"2024-02-03T03:36:16.745723Z","shell.execute_reply":"2024-02-03T03:36:16.931970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = tf.reshape(prediction, -1).numpy()\nprediction","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:19.887460Z","iopub.execute_input":"2024-02-03T03:36:19.888459Z","iopub.status.idle":"2024-02-03T03:36:19.895905Z","shell.execute_reply.started":"2024-02-03T03:36:19.888423Z","shell.execute_reply":"2024-02-03T03:36:19.894973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.sum()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:22.414666Z","iopub.execute_input":"2024-02-03T03:36:22.415088Z","iopub.status.idle":"2024-02-03T03:36:22.422610Z","shell.execute_reply.started":"2024-02-03T03:36:22.415053Z","shell.execute_reply":"2024-02-03T03:36:22.421474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:34.619596Z","iopub.execute_input":"2024-02-03T03:36:34.620228Z","iopub.status.idle":"2024-02-03T03:36:34.628303Z","shell.execute_reply.started":"2024-02-03T03:36:34.620196Z","shell.execute_reply":"2024-02-03T03:36:34.627394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[TARGET_COLUMNS] = prediction\nsub_df","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:37.870723Z","iopub.execute_input":"2024-02-03T03:36:37.871099Z","iopub.status.idle":"2024-02-03T03:36:37.886798Z","shell.execute_reply.started":"2024-02-03T03:36:37.871067Z","shell.execute_reply":"2024-02-03T03:36:37.885728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T03:36:47.924053Z","iopub.execute_input":"2024-02-03T03:36:47.924431Z","iopub.status.idle":"2024-02-03T03:36:47.933980Z","shell.execute_reply.started":"2024-02-03T03:36:47.924399Z","shell.execute_reply":"2024-02-03T03:36:47.933080Z"},"trusted":true},"execution_count":null,"outputs":[]}]}