{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"\nCreated on 2024/04/02\n\n@author: fred4code\n\nnotebook for the Kaggle competition:\n/kaggle/input/hms-harmful-brain-activity-classification\n\"\"\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A Multi-Input Model made of LSTM and CNN neural networks","metadata":{}},{"cell_type":"markdown","source":"This is a Multi-Input Model that uses a mix of LSTMs and CNNs.\n\nLSTMs are used for the EEG signals and the spectrums data.\n\nCNNs are used for the spectrum data only.\n\nThe EEG signal used for the train are 10 seconds sampled at 200 samples per second, calculated with the offsets.\n\nThe spectrum data have been calculated directly from the EEG signal used for the train.","metadata":{}},{"cell_type":"markdown","source":"# Check tensorflow and keras versions. This notebook has been tested with tensorflow version 2.15.0 and keras version 3.0.5","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Versione TensorFlow:\", tf.__version__)\n\nimport keras\nprint(\"Keras version:\", keras.__version__)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions used to support the data analysis","metadata":{}},{"cell_type":"code","source":"from types import MethodType\n\nclass MKaggleEEG:\n\n    def __init__(self):\n                \n        self.MV_Save_Also = False\n        \n        self.MV_input_directory_path = None\n        self.MV_ouptut_directory_path = None     \n        self.MV_output_subdirectory_name = None\n        \n        self.MV_file_list_eeg_id = None\n        self.MV_df_electrodes_couples = None       \n        self.MV_train = None\n        self.MV_file_list_eeg_id_int = None\n        self.MV_df_train_normalized = None\n        self.MV_df_list_subsamples_eeg_s10 = None\n        \n        self.MV_file_list_eeg_id_egg_sub_id_s10 = None\n        self.MV_df_list_subsamples_eeg_s10 = None\n        \n        self.MV_list_df_subsamples_eeg_differences = None\n        self.MV_train_Y = None\n        self.MV_train_X= None\n\n        self.MV_n_train_balanced = None\n        self.MV_batch_size = None\n        self.MV_epochs = None\n        self.MV_train_balanced = None\n        self.MM_split_percentage = None\n\n        pass\n\nm = MKaggleEEG()\n\ndef MM_File_List_Get(self, pDirectoryBase, pDirectorySub, pFileType):\n    \n    import os\n\n    full_path = os.path.join(pDirectoryBase, pDirectorySub)\n    \n    if not os.path.exists(full_path):\n        raise FileNotFoundError(f'Path specified: {full_path} does not exist.')\n    \n    files_without_extension = [\n        os.path.splitext(file)[0] for file in os.listdir(full_path) if file.endswith('.' + pFileType)\n    ]\n    \n    files_without_extension.sort()\n    \n    return files_without_extension\nm.MM_File_List_Get = MethodType(MM_File_List_Get, m)\n\n\ndef MM_EEG_Electrodes_Couples_Get(self, pElectrodesCouples):\n    \n   \n    import pandas as pd\n\n    data = None\n    \n    if pElectrodesCouples == 18:\n        data = {\n            \"Electrode_1\": [\"Fp1\", \"F7\", \"T3\", \"T5\", \"Fp2\", \"F8\", \"T4\", \"T6\", \"Fp1\", \"F3\", \"C3\", \"P3\", \"Fp2\", \"F4\", \"C4\", \"P4\", \"Fz\", \"Cz\"],\n            \"Electrode_2\": [\"F7\", \"T3\", \"T5\", \"O1\", \"F8\", \"T4\", \"T6\", \"O2\", \"F3\", \"C3\", \"P3\", \"O1\", \"F4\", \"C4\", \"P4\", \"O2\", \"Cz\", \"Pz\"]\n        }\n    elif pElectrodesCouples == 8:\n\n        data = {\n            \"Electrode_1\": ['Fp1', 'T3', 'Fp1', 'C3', 'Fp2', 'C4', 'Fp2', 'T4'],\n            \"Electrode_2\": ['T3', 'O1', 'C3', 'O1', 'C4', 'O2', 'T4', 'O2']\n        }\n        \n   \n    df = pd.DataFrame(data)\n\n    return df\nm.MM_EEG_Electrodes_Couples_Get = MethodType(MM_EEG_Electrodes_Couples_Get, m) \n    \n\ndef MM_Train_Get(self, pDirectoryBase):\n    import pandas as pd\n    import os\n    \n    \n    file_train_path = os.path.join(pDirectoryBase, 'train.csv')\n    \n    df_train = pd.read_csv(file_train_path)\n    \n    return df_train\nm.MM_Train_Get = MethodType(MM_Train_Get, m)    \n\n\n\ndef MM_Train_Normalizing_Classes_to_Binary_Classification_Add(self, pDf):\n    \n    import pandas as pd\n    \n    df_train = pDf.copy()\n\n    for col in ['seizure_is', 'lpd_is', 'gpd_is', 'lrda_is', 'grda_is', 'other_is']:\n        df_train[col] = 0\n    \n    def update_rows(row):\n        if row['expert_consensus'] == 'Seizure':\n            row['seizure_is'] = 1\n        elif row['expert_consensus'] == 'LPD':\n            row['lpd_is'] = 1\n        elif row['expert_consensus'] == 'GPD':\n            row['gpd_is'] = 1\n        elif row['expert_consensus'] == 'LRDA':\n            row['lrda_is'] = 1\n        elif row['expert_consensus'] == 'GRDA':\n            row['grda_is'] = 1\n        elif row['expert_consensus'] == 'Other':\n            row['other_is'] = 1\n        return row\n    \n    df_train = df_train.apply(update_rows, axis=1)\n    \n    return df_train     \nm.MM_Train_Normalizing_Classes_to_Binary_Classification_Add = MethodType(MM_Train_Normalizing_Classes_to_Binary_Classification_Add, m)   \n\n\n\ndef MM_Parquet_Files_Process(self, df, eeg_id, eeg_sub_id, electrode_pairs_df,pTrainNormalized, base_directory, output_subdirectory):\n    import os\n    import pandas as pd\n    \n    output_directory = os.path.join(base_directory, output_subdirectory)\n\n    if not os.path.exists(output_directory):\n        os.makedirs(output_directory)\n    \n    differences_df = pd.DataFrame()                  \n    for index, row in electrode_pairs_df.iterrows():\n        electrode_1 = row['Electrode_1']\n        electrode_2 = row['Electrode_2']\n                      \n        column_name = f'{electrode_1}-{electrode_2}'\n        differences_df[column_name] = df[electrode_1] - df[electrode_2]\n    if 'EKG' in df.columns:\n        differences_df['EKG'] = df['EKG']\n    \n   \n    train_Y = pTrainNormalized[(pTrainNormalized['eeg_id'] == int(eeg_id)) & (pTrainNormalized['eeg_sub_id'] == int(eeg_sub_id))]\n\n    def MM_Dataframe_To_Train_Y(df, pColumnsType):\n        \n        import numpy as np\n        \n        y_train = None\n        if pColumnsType == \"prob\":\n            prob_columns = [col for col in df.columns if col.endswith('_prob')]\n            y_train = df[prob_columns].values\n        elif pColumnsType == \"expert_consensus_encoded\":\n            y_train = df['expert_consensus_encoded'].values\n            y_train = np.expand_dims(y_train, axis=-1)\n        elif pColumnsType == \"one_hot\":\n            prob_columns = [col for col in df.columns if col.endswith('_is')]\n            y_train = df[prob_columns].values\n              \n        return y_train\n    train_Y = MM_Dataframe_To_Train_Y(train_Y, 'one_hot')\n\n\n    output_filename = f\"{eeg_id}_{eeg_sub_id}_e.hdf5\"\n    \n    output_file_path = os.path.join(output_directory, output_filename)\n\n    differences_df_np = differences_df.to_numpy()\n    import h5py\n    f = h5py.File(output_file_path, 'w')\n\n    f.create_dataset('X', data=differences_df_np)\n    f.create_dataset('Y', data=train_Y)\n\nm.MM_Parquet_Files_Process = MethodType(MM_Parquet_Files_Process, m) \n\ndef MM_EEG_Subsamples_From_Parquet_Get(self, eeg_id, eeg_sub_id, pDirectoryBase, pDirectoryOutput, pDfTrain, pLengthSeconds=10, pStartSecondAfterTheOffset=20, sample_rate = 200):\n    \n\n    import pandas as pd\n    import os\n    \n    eeg_file_path = os.path.join(pDirectoryBase, 'train_eegs', f'{eeg_id}.parquet')\n    df_eeg = pd.read_parquet(eeg_file_path)\n\n    df_train_filtered = pDfTrain[(pDfTrain['eeg_id'] == eeg_id) & (pDfTrain['eeg_sub_id'] == eeg_sub_id)]\n\n    list_df_subsamples = []\n    \n    for index, row in df_train_filtered.iterrows():\n\n        index_start = int((row['eeg_label_offset_seconds'] + pStartSecondAfterTheOffset) * sample_rate)\n        index_end = index_start + (pLengthSeconds * sample_rate) \n            \n        df_subsample = df_eeg.iloc[index_start:index_end].copy()\n        \n        df_subsample = df_subsample.reset_index(drop=True)\n\n        list_df_subsamples.append(df_subsample)\n\n    return list_df_subsamples\nm.MM_EEG_Subsamples_From_Parquet_Get = MethodType(MM_EEG_Subsamples_From_Parquet_Get, m) \n\n\ndef MM_EEG_Df_List_Get(self, pListFileParquet, pDirectoryOutput, pDirectorySub, pFileNameAdded):\n    \n    import pandas as pd\n    import os\n    \n    list_df = []\n    \n    for file_name in pListFileParquet:\n        file_path = os.path.join(pDirectoryOutput, pDirectorySub, f\"{file_name}{pFileNameAdded}.parquet\")\n                  \n        df = pd.read_parquet(file_path)\n        \n        list_df.append(df)\n    \n    return list_df  \nm.MM_EEG_Df_List_Get = MethodType(MM_EEG_Df_List_Get, m)   \n\n\ndef MM_Dataframe_To_Train_Y(self, df, pColumnsType):\n    \n    import numpy as np\n    \n    prob_columns = None\n    y_train = None\n    if pColumnsType == \"prob\":\n        prob_columns = [col for col in df.columns if col.endswith('_prob')]\n        y_train = df[prob_columns].values\n    elif pColumnsType == \"expert_consensus_encoded\":\n        y_train = df['expert_consensus_encoded'].values\n        y_train = np.expand_dims(y_train, axis=-1)\n    elif pColumnsType == \"one_hot\":\n        prob_columns = [col for col in df.columns if col.endswith('_is')]\n        y_train = df[prob_columns].values\n        \n        \n    return y_train, prob_columns\nm.MM_Dataframe_To_Train_Y = MethodType(MM_Dataframe_To_Train_Y, m) \n\n\ndef MM_LSTM_Dataframe_Features_To_Train_X(self, pDf):\n    \n    train_X = pDf.values.reshape(1, pDf.shape[0], pDf.shape[1])\n    \n\n    return train_X\nm.MM_LSTM_Dataframe_Features_To_Train_X = MethodType(MM_LSTM_Dataframe_Features_To_Train_X, m) \n\ndef MM_LSTM_List_Dataframe_To_Train_X(self, pListDf):\n    \n\n    import numpy as np\n    \n    transformed_arrays = []\n    \n    for df in pListDf:\n        transformed_array = self.MM_LSTM_Dataframe_Features_To_Train_X(df)\n        transformed_arrays.append(transformed_array)\n    \n    train_X_complessivo = np.concatenate(transformed_arrays, axis=0)\n    \n    return train_X_complessivo\nm.MM_LSTM_List_Dataframe_To_Train_X = MethodType(MM_LSTM_List_Dataframe_To_Train_X, m)    \n\ndef MM_Classes_Balancing_Set(self, df, pColumnLabel='expert_consensus', pColumnId='eeg_id', random_state=42):\n    import pandas as pd\n    gruppi = df.groupby(pColumnLabel)\n    value_min = gruppi[pColumnId].nunique().min()\n    \n    df_balanced = pd.DataFrame()\n    \n    for classe in df[pColumnLabel].unique():\n        df_class = df[df[pColumnLabel] == classe]\n        \n        df_class_unique = df_class.groupby(pColumnId).sample(n=1, random_state=random_state)\n        \n        df_class_sample = df_class_unique.sample(n=min(len(df_class_unique), value_min), random_state=random_state)\n        \n        df_balanced = pd.concat([df_balanced, df_class_sample], ignore_index=True)\n    \n    return df_balanced\nm.MM_Classes_Balancing_Set = MethodType(MM_Classes_Balancing_Set, m)  \n\ndef MM_Class_Order_Rows(self, pDfBalanced, pColumnLabel='expert_consensus'):\n    import pandas as pd\n    \n    pDfBalanced = pDfBalanced.sort_values(by=pColumnLabel)\n    \n    class_unique = pDfBalanced[pColumnLabel].unique()\n    \n    df_parts = []\n    \n    righe_per_classe = pDfBalanced[pColumnLabel].value_counts().iloc[0]\n    \n    for i in range(righe_per_classe):\n        for c_u in class_unique:\n            row = pDfBalanced[pDfBalanced[pColumnLabel] == c_u].iloc[i]\n            df_parts.append(row)\n    \n    df_ordered = pd.DataFrame(df_parts).reset_index(drop=True)\n    \n    return df_ordered  \nm.MM_Class_Order_Rows = MethodType(MM_Class_Order_Rows, m)  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data analysis...data loading, normalization, model training, validation, prediction and submission","metadata":{}},{"cell_type":"code","source":"def MM_Main_1(self):\n    \n    import pandas as pd\n    pd.set_option('display.max_columns', None)\n    pd.set_option('display.max_rows', None)\n              \n    print(\">set this to True to extract 10 seconds of eeg samples with the frequency of 200 sample per second.\")\n    print(\"Set this to False if the samples have already been extracted\")\n    self.MV_Save_Also = True\n    \n    print(\">set training parameters\")\n    self.MV_n_train_balanced = 1\n    self.MV_batch_size = 32\n    self.MV_epochs = 50\n    self.MM_split_percentage = 0.7\n    \n    print('>get the names of all the train files')\n    self.MV_input_directory_path= '/kaggle/input/hms-harmful-brain-activity-classification'\n    self.MV_ouptut_directory_path = '/kaggle/working'\n    self.MV_output_subdirectory_name = 'm_train_seconds_10_e_test_differences'\n    self.MV_file_list_eeg_id = self.MM_File_List_Get(self.MV_input_directory_path, 'train_eegs', 'parquet')\n\n    print('>get electrodes couples as in the samples pdfs')\n    self.MV_df_electrodes_couples = self.MM_EEG_Electrodes_Couples_Get(18)\n\n\n    print('>get train.csv')\n    self.MV_train = self.MM_Train_Get(self.MV_input_directory_path)\n    print('>be sure to take all the files that are listed in train.csv')\n    self.MV_file_list_eeg_id_int = [int(x) for x in self.MV_file_list_eeg_id]\n    self.MV_train = self.MV_train[self.MV_train['eeg_id'].isin(self.MV_file_list_eeg_id_int)]\n\n\n    print('>normalize train to have a one-hot encoding classification')\n    self.MV_df_train_normalized = self.MM_Train_Normalizing_Classes_to_Binary_Classification_Add(self.MV_train)\n    \n    print('>add a column with the name of the file that for each sample will contain parquet files elaborations')\n    self.MV_df_train_normalized['file_name'] = self.MV_df_train_normalized.apply(lambda x: str(x['eeg_id']) + \"_\" + str(x['eeg_sub_id']) + \"_e.hdf5\", axis=1)\n\n    print(\">balanced the classes discarding samples...it make also the process of training faster\")\n    self.MV_train_balanced = self.MM_Classes_Balancing_Set(self.MV_df_train_normalized)\n\n    print(\">reorder the balanced samples\")\n    self.MV_train_balanced = self.MM_Class_Order_Rows(self.MV_train_balanced)\n    \n\n    print('>save 10 seconds of EEG with a frequency of 200 samples per second')\n    if (self.MV_Save_Also):\n        for index, row in self.MV_train_balanced.iterrows():\n            eeg_id = row['eeg_id']\n            eeg_sub_id = row['eeg_sub_id']\n            list_df_subsamples_eeg_n = self.MM_EEG_Subsamples_From_Parquet_Get(int(eeg_id), int(eeg_sub_id), self.MV_input_directory_path, self.MV_ouptut_directory_path, self.MV_train_balanced, 10, 20) \n            for index, df in enumerate(list_df_subsamples_eeg_n):    \n                self.MM_Parquet_Files_Process(df, eeg_id, eeg_sub_id, self.MV_df_electrodes_couples, self.MV_train_balanced, self.MV_ouptut_directory_path, self.MV_output_subdirectory_name)  \n    print('10 seconds samples saved!')   \n\n\n    from tensorflow.keras.utils import Sequence\n    class HDF5DataGenerator(Sequence):\n        \n        def MM_EEG_Filtering_MovingAverage_Df(self, pDf, pWindowSize):\n            import numpy as np\n\n            \n            df_copy = pDf.copy() \n        \n            def mm_moving_average(signal, window_size):\n                window = np.ones(int(window_size)) / float(window_size)\n                return np.convolve(signal, window, 'same')            \n            \n            return df_copy.apply(lambda col: mm_moving_average(col, pWindowSize))\n        \n        def MM_DataFrame_Spectrogram_Filtered_Get(self, df, fs, mains_f=60., filter_mains=True, filter_DC=True, normalize_signals=False):\n            \n            import numpy as np\n            import pandas as pd\n            from scipy.signal import spectrogram, butter, lfilter\n        \n            df_copy = df.copy()\n        \n            for col in df_copy.columns:\n                signal = df_copy[col].values\n                \n                if normalize_signals:\n                    signal = (signal - np.mean(signal)) / np.std(signal)\n        \n                if filter_DC:\n                    signal = signal - np.mean(signal)\n        \n                if filter_mains:\n     \n                    b, a = butter(2, [(mains_f-1)/(fs/2), (mains_f+1)/(fs/2)], btype='bandstop', output='ba')\n                    signal = lfilter(b, a, signal)\n        \n                df_copy[col] = signal\n                \n      \n            first_column_name = df_copy.columns[0]\n            f, t, Sxx = spectrogram(df_copy[first_column_name].values, fs, nperseg=None, noverlap=None)\n\n            return pd.DataFrame(Sxx)\n        \n        \n        def __init__(self, pTrain, pColumns, pDirectory, pBatchSize, pMode, pSplitPercentage):\n            \n            import os\n            \n            self.file_paths = []  \n            self.batch_size = pBatchSize\n            \n            self.directory = pDirectory\n            self.mode = pMode\n            self.channels = pColumns\n            self.splitPercentage = pSplitPercentage\n    \n\n            if pTrain is not None:\n                file_names = set(pTrain['file_name'])\n                self.file_paths = [os.path.join(self.directory, f) for f in file_names]\n            else:\n                self.file_paths = [os.path.join(self.directory, f) for f in os.listdir(self.directory) if f.endswith('.hdf5')]\n\n            if self.splitPercentage is not None:\n                split_index = int(len(self.file_paths) * pSplitPercentage)\n            \n                if self.mode == 'train':\n                    self.file_paths = self.file_paths[:split_index]\n                elif self.mode == 'validation':\n                    self.file_paths = self.file_paths[split_index:]\n                else:\n                    raise ValueError(\"mode parameter must be 'train' or 'validation'.\")\n\n                \n            self.num_samples = len(self.file_paths)\n            \n            \n            \n        def __len__(self):\n            \n            import numpy as np\n            return np.ceil(len(self.file_paths) / self.batch_size).astype(int)\n        \n        def __getitem__(self, index):\n            import h5py\n            import pandas as pd\n            import numpy as np\n            import os\n            \n            start_idx = index * self.batch_size\n            end_idx = start_idx + self.batch_size\n    \n            batch_input1 = []\n            batch_input2 = []\n            batch_labels = []\n            eeg_id = None\n    \n            for file_path in self.file_paths[start_idx:end_idx]:\n                with h5py.File(file_path, 'r') as f:\n                    \n\n                    input1 = f['X'][()] \n                    if self.splitPercentage is not None:\n                        input1 = pd.DataFrame(input1) \n                        \n                        labels = f['Y'][()]\n                        batch_labels.append(labels)\n                \n                    else:\n                   \n                        input1 = pd.DataFrame(input1)\n                        start_index = len(input1) // 2 - 1000\n                        end_index = start_index + 2000\n                        start_index = max(start_index, 0)\n                        input1 = input1.iloc[start_index:end_index]\n                        input1 = input1.reset_index(drop=True)\n                        \n                        file_name_with_extension = os.path.basename(file_path)\n                        file_name_without_extension = os.path.splitext(file_name_with_extension)[0]\n                        eeg_id = np.int64(file_name_without_extension)\n\n                    input1 = self.MM_EEG_Filtering_MovingAverage_Df( input1, 10)\n\n                    input1 = input1.fillna(0)\n                    batch_input1.append(input1)\n                    \n              \n                    list_df_spectrograms = []\n                    for col in input1.columns:\n                        df_spectrogram = self.MM_DataFrame_Spectrogram_Filtered_Get(input1.iloc[:, [col]], 200)\n                        list_df_spectrograms.append(df_spectrogram)\n                    list_df_spettrogramms_index_reset = [df.reset_index(drop=True) for df in list_df_spectrograms]\n                    df_spectrograms_concat = pd.concat(list_df_spettrogramms_index_reset, axis=1)\n                    batch_input2.append(df_spectrograms_concat)\n\n\n            \n            batch_input1 = np.stack(batch_input1, axis=0)\n            batch_input2 = np.stack(batch_input2, axis=0)\n            \n            if (self.splitPercentage != None):\n                \n                batch_labels = np.squeeze(batch_labels)\n                batch_labels = np.stack(batch_labels, axis=0)\n                return [batch_input1, batch_input2], batch_labels\n\n            else:\n                \n                batch_labels = np.zeros((1,6), dtype=int)\n                batch_labels = np.stack(batch_labels, axis=0)\n                return [batch_input1, batch_input2], batch_labels, eeg_id\n                \n            \n        def MM_Generator_Train_Validation(self):\n            import numpy as np\n            for index in range(len(self)):\n                batch = self.__getitem__(index)\n                \n\n                for example in range(batch[0][0].shape[0]): \n            \n                    batch_input1 = batch[0][0][example]\n                    batch_input2 = batch[0][1][example]\n                    batch_input2_cnn_np = np.expand_dims(batch_input2, axis=-1)\n        \n                    batch_labels = batch[1][example]\n                    \n                    yield (batch_input1, batch_input2, batch_input2_cnn_np), batch_labels\n                    \n        def MM_Generator_Prediction(self):\n            import numpy as np\n            for index in range(len(self)):\n                batch = self.__getitem__(index)\n\n                    \n                for example in range(batch[0][0].shape[0]): \n            \n                    batch_input1 = batch[0][0][example]\n                    batch_input2 = batch[0][1][example]\n                    batch_input2_cnn_np = np.expand_dims(batch_input2, axis=-1)\n        \n                    batch_labels = batch[1][example]\n                    \n                    eeg_id = batch[2]\n                    \n                    yield (batch_input1, batch_input2, batch_input2_cnn_np), batch_labels, eeg_id\n                    \n\n    def MM_Training_Set(pTrain, pRowReducingSetFactor, pBatchSize, pEpochs, pSplitPercentage, pOutputDirectoryPath, pOutputSubDirectoryName):              \n        import os\n        import tensorflow as tf\n        from tensorflow.keras.models import Model\n        from tensorflow.keras.layers import Input, LSTM, Dropout, Dense, concatenate\n        from tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten\n        from tensorflow.keras.layers import Conv2D, MaxPooling2D\n        from tensorflow.keras.optimizers import Adam\n\n\n        train_path = os.path.join(pOutputDirectoryPath, pOutputSubDirectoryName) \n\n        num_rows = len(pTrain) // pRowReducingSetFactor\n        pTrain = pTrain.iloc[:num_rows]\n        \n        #\n        total_training_samples = int(pTrain.shape[0] * pSplitPercentage)\n        total_validation_samples = int(pTrain.shape[0] * (1-pSplitPercentage))\n        train_steps_per_epochs = total_training_samples // pBatchSize\n        validation_steps = total_validation_samples // pBatchSize\n\n        train_generator = HDF5DataGenerator(pTrain, 19, pDirectory=train_path, pBatchSize=pBatchSize, pMode ='train', pSplitPercentage=pSplitPercentage)\n\n        validation_generator = HDF5DataGenerator(pTrain, 19, pDirectory=train_path, pBatchSize=pBatchSize, pMode = 'validation', pSplitPercentage=pSplitPercentage)\n\n\n        batch_input1 = train_generator.__getitem__(0)[0][0][0]\n        batch_input2 = train_generator.__getitem__(0)[0][1][0]\n        batch_labels = train_generator.__getitem__(0)[1][0]\n\n        input1 = Input(shape=(batch_input1.shape[0], batch_input1.shape[1]), name='input1')\n        input2 = Input(shape=(batch_input2.shape[0], batch_input2.shape[1],), name='input2')\n        input2_cnn2 = Input(shape=(batch_input2.shape[0], batch_input2.shape[1],1), name='input2_cnn2')\n\n        lstm_1_1 = LSTM(64, return_sequences=True)(input1)\n        dropout_1_1 = Dropout(0.2)(lstm_1_1)\n        lstm_1_2 = LSTM(64, return_sequences=False)(dropout_1_1)\n        dropout_1_2 = Dropout(0.2)(lstm_1_2)\n     \n        lstm_2_1 = LSTM(64, return_sequences=True)(input2)\n        dropout_2_1 = Dropout(0.2)(lstm_2_1)\n        lstm_2_2 = LSTM(64, return_sequences=False)(dropout_2_1)\n        dropout_2_2 = Dropout(0.2)(lstm_2_2)\n \n        cnn_1 = Conv1D(filters=32, kernel_size=3, activation='relu')(input2)\n        cnn_1 = MaxPooling1D(pool_size=2)(cnn_1)\n        cnn_1 = Conv1D(filters=64, kernel_size=3, activation='relu')(cnn_1)\n        cnn_1 = MaxPooling1D(pool_size=2)(cnn_1)\n        cnn_1 = Flatten()(cnn_1)\n\n        cnn_2 = Conv2D(filters=32, kernel_size=(3, 3), activation='relu')(input2_cnn2)\n        cnn_2 = MaxPooling2D(pool_size=(2, 2))(cnn_2)\n        cnn_2 = Conv2D(filters=64, kernel_size=(3, 3), activation='relu')(cnn_2)\n        cnn_2 = MaxPooling2D(pool_size=(2, 2), padding='same')(cnn_2)\n        cnn_2 = Flatten()(cnn_2)\n\n\n        merged = concatenate([dropout_1_2, dropout_2_2, cnn_1, cnn_2])\n\n        outputs = Dense(batch_labels.shape[0], activation='softmax')(merged)\n\n        model = Model(inputs=[input1, input2, input2_cnn2], outputs=outputs)\n\n        optimizer = Adam(learning_rate=0.001)\n        model.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=['accuracy'])\n\n       \n        dataset_train = tf.data.Dataset.from_generator(\n            lambda: train_generator.MM_Generator_Train_Validation(),\n            output_signature=(\n                (\n                    tf.TensorSpec(shape=(batch_input1.shape[0], batch_input1.shape[1]), dtype=tf.float32), \n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1]), dtype=tf.float32),  \n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1], 1), dtype=tf.float32) \n                ),\n                tf.TensorSpec(shape=(batch_labels.shape[0],), dtype=tf.int32) \n            )\n        )\n        \n\n        dataset_train_batched = dataset_train.batch(pBatchSize)\n        dataset_train_batched = dataset_train_batched.repeat()\n        \n        validation_dataaset = tf.data.Dataset.from_generator(\n            lambda: validation_generator.MM_Generator_Train_Validation(),\n            output_signature=(\n                (\n                    tf.TensorSpec(shape=(batch_input1.shape[0], batch_input1.shape[1]), dtype=tf.float32), \n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1]), dtype=tf.float32), \n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1], 1), dtype=tf.float32)\n                ),\n                tf.TensorSpec(shape=(batch_labels.shape[0],), dtype=tf.int32)  \n            )\n        )\n\n        validation_dataset_batched = validation_dataaset.batch(pBatchSize)\n        validation_dataset_batched = validation_dataset_batched.repeat()\n\n        history = model.fit(dataset_train_batched,\n            epochs=pEpochs,\n            steps_per_epoch=train_steps_per_epochs,\n            validation_data=validation_dataset_batched, \n            validation_steps=validation_steps\n            )\n\n        return model, history\n    \n    print(\">train the model\")\n    print(\"get 1/n of the balanced sample...to make the process faster. It can be set to n=1 for all the balanced set\")\n    model, history = MM_Training_Set(self.MV_train_balanced, self.MV_n_train_balanced, self.MV_batch_size, self.MV_epochs, self.MM_split_percentage, self.MV_ouptut_directory_path, self.MV_output_subdirectory_name)    \n    \n\n    def MM_Prediction_Set(pModel,pBatchSize,pTrain, pInputDirectoryPath, pOutputDirectoryPath):\n        import os\n        import pandas as pd\n        import tensorflow as tf\n        \n        output_file_path = os.path.join(pOutputDirectoryPath,\"test_eegs\")\n        if not os.path.exists(output_file_path):\n            os.makedirs(output_file_path)\n    \n        column_names = [col for col in pTrain.columns if col.endswith('_is')]\n        column_names = [col.replace('_is', '_vote') for col in column_names]\n        \n        df_test_csv = pd.read_csv(os.path.join(pInputDirectoryPath,\"test.csv\"))\n\n        for index, row in df_test_csv.iterrows(): \n            test_eeg_id = str(row[\"eeg_id\"])   \n            df = pd.read_parquet(os.path.join(pInputDirectoryPath,\"test_eegs\", test_eeg_id + \".parquet\"))\n            differences_df = pd.DataFrame()                  \n            for index, row in self.MV_df_electrodes_couples.iterrows():\n                electrode_1 = row['Electrode_1']\n                electrode_2 = row['Electrode_2']\n                column_name = f'{electrode_1}-{electrode_2}'\n                differences_df[column_name] = df[electrode_1] - df[electrode_2]\n            if 'EKG' in df.columns:\n                differences_df['EKG'] = df['EKG']\n                \n            output_file_path = os.path.join(pOutputDirectoryPath,\"test_eegs\",test_eeg_id + \".hdf5\")\n            differences_df_np = differences_df.to_numpy()\n            import h5py\n            f = h5py.File(output_file_path, 'w')\n            f.create_dataset('X', data=differences_df_np)\n            f.create_dataset('Y', data=[])\n            f.close()\n            \n\n        prediction_generator = HDF5DataGenerator(None, 19, pDirectory=os.path.join(pOutputDirectoryPath,\"test_eegs\"), pBatchSize=pBatchSize, pMode = None, pSplitPercentage=None)\n\n        batch_input1 = prediction_generator.__getitem__(0)[0][0][0]\n        batch_input2 = prediction_generator.__getitem__(0)[0][1][0]\n        batch_labels = prediction_generator.__getitem__(0)[1][0]\n\n        prediction_dataset = tf.data.Dataset.from_generator(\n            lambda: prediction_generator.MM_Generator_Prediction(),\n            output_signature=(\n                (\n                    tf.TensorSpec(shape=(batch_input1.shape[0], batch_input1.shape[1]), dtype=tf.float32), \n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1]), dtype=tf.float32),\n                    tf.TensorSpec(shape=(batch_input2.shape[0], batch_input2.shape[1], 1), dtype=tf.float32)\n                ),\n                tf.TensorSpec(shape=(batch_labels.shape[0],), dtype=tf.int32),\n                tf.TensorSpec(shape=(), dtype=tf.int64)\n            )\n        )\n\n        predictions = []\n\n        df_predictions = pd.DataFrame()\n        \n        prediction_dataset_batched = prediction_dataset.batch(pBatchSize)\n        for data_batched in prediction_dataset_batched:\n            \n            x = data_batched[0]\n            eeg_id = data_batched[2]\n            \n            prediction = model.predict(x)\n            predictions.append(prediction)\n\n            \n            df_temp = pd.DataFrame(prediction.reshape(1, -1), columns=column_names)\n            df_temp['eeg_id'] = eeg_id\n            \n            df_predictions = pd.concat([df_predictions, df_temp], ignore_index=True)\n\n        df_predictions = pd.merge(df_test_csv, df_predictions, on='eeg_id', how='left')\n        \n        columns_to_select = [\"eeg_id\", \"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\n        df_predictions = df_predictions[columns_to_select]\n\n        import shutil\n        shutil.rmtree('/kaggle/working/m_train_seconds_10_e_test_differences')\n        shutil.rmtree('/kaggle/working/test_eegs')\n        \n        df_predictions.to_csv('submission.csv', index=False)\n        df_predictions.to_csv(os.path.join(pOutputDirectoryPath,\"submission.csv\"), index=False)\n        \n        pModel.save('model.h5')\n        pModel.save(os.path.join(pOutputDirectoryPath,\"model.h5\"))\n\n        print(\">Check the probability sum to one in each row of the submission dataframe\")\n        print(df_predictions.iloc[:,-6:].sum(axis=1))\n        \n        from IPython.display import display\n        print(\">submission values\")\n        display(df_predictions)\n\n\n    print(\">predict the values\")   \n    print(\"take test.csv and produce sample_submission.csv\")    \n    MM_Prediction_Set(model, 1, self.MV_train_balanced, self.MV_input_directory_path, self.MV_ouptut_directory_path)\n    print(\"sample_submission.csv done!\")  \n\n    def MM_Plot_Accuracy_Loss(pHistory):\n\n        import matplotlib.pyplot as plt\n    \n        plt.figure(figsize=(16, 5))\n        \n        plt.subplot(1, 2, 1) \n        plt.plot(pHistory.history['loss'], label='Train Loss')\n        plt.plot(pHistory.history['val_loss'], label='Validation Loss')\n        plt.title('Model Loss')\n        plt.ylabel('Loss')\n        plt.xlabel('Epoch')\n        plt.legend()\n        \n        plt.subplot(1, 2, 2) \n        plt.plot(pHistory.history['accuracy'], label='Train Accuracy')\n        plt.plot(pHistory.history['val_accuracy'], label='Validation Accuracy')\n        plt.title('Model Accuracy')\n        plt.ylabel('Accuracy')\n        plt.xlabel('Epoch')\n        plt.legend()\n        \n        plt.show()\n        \n    print(\">plot accuracy and loss\")   \n    MM_Plot_Accuracy_Loss(history)\n\nm.MM_Main_1 = MethodType(MM_Main_1, m)\n\nm.MM_Main_1()","metadata":{},"execution_count":null,"outputs":[]}]}