{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":59093,"databundleVersionId":7469972},{"sourceType":"datasetVersion","sourceId":7754261,"datasetId":4532886,"databundleVersionId":7854993},{"sourceType":"datasetVersion","sourceId":7392733,"datasetId":4297749,"databundleVersionId":7483738},{"sourceType":"datasetVersion","sourceId":7979348,"datasetId":4673903,"databundleVersionId":8089892},{"sourceType":"datasetVersion","sourceId":2378330,"datasetId":492658,"databundleVersionId":2420211},{"sourceType":"datasetVersion","sourceId":7402356,"datasetId":4304475,"databundleVersionId":7493457},{"sourceType":"datasetVersion","sourceId":7753795,"datasetId":4533761,"databundleVersionId":7854515},{"sourceType":"kernelVersion","sourceId":158958765}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TRAINING","metadata":{}},{"cell_type":"markdown","source":"## Global constants","metadata":{}},{"cell_type":"code","source":"# Set to True for inference only, False for training\nONLY_INFERENCE = True\n\n# Configuration for model training\nFOLDS = 5\nEPOCHS = 4\nBATCH = 32\nNAME = 'None'\n\nSPEC_SIZE  = (512, 512, 3)\nCLASSES = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nN_CLASSES = len(CLASSES)\nTARGETS = CLASSES\nimport matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:40:44.607834Z","iopub.execute_input":"2024-03-30T05:40:44.608301Z","iopub.status.idle":"2024-03-30T05:40:44.621459Z","shell.execute_reply.started":"2024-03-30T05:40:44.608260Z","shell.execute_reply":"2024-03-30T05:40:44.620479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"IMPORTS","metadata":{}},{"cell_type":"code","source":"!pip install --no-index --find-links=/kaggle/input/tf-efficientnet-whl-files /kaggle/input/tf-efficientnet-whl-files/efficientnet-1.1.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:40:44.623219Z","iopub.execute_input":"2024-03-30T05:40:44.623567Z","iopub.status.idle":"2024-03-30T05:40:58.654160Z","shell.execute_reply.started":"2024-03-30T05:40:44.623535Z","shell.execute_reply":"2024-03-30T05:40:58.653258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\nimport random\nimport sys\nimport time\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom tensorflow.keras import backend as K\nfrom tqdm import tqdm \nfrom scipy.ndimage import gaussian_filter\nfrom scipy.signal import butter, filtfilt, iirnotch\nfrom scipy.signal import spectrogram as spectrogram_np\n\nimport efficientnet.tfkeras as efn\n\nsys.path.append(f'/kaggle/input/kaggle-kl-div')\nfrom kaggle_kl_div import score","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:40:58.655727Z","iopub.execute_input":"2024-03-30T05:40:58.656107Z","iopub.status.idle":"2024-03-30T05:41:19.597858Z","shell.execute_reply.started":"2024-03-30T05:40:58.656072Z","shell.execute_reply":"2024-03-30T05:41:19.596870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!nvidia-smi\n\n# Installation of RAPIDS to Use cuSignal\n!cp ../input/rapids/rapids.0.17.0 /opt/conda/envs/rapids.tar.gz\n!cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n!rm /opt/conda/envs/rapids.tar.gz\n\nsys.path += [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"]\nsys.path += [\"/opt/conda/envs/rapids/lib/python3.7\"]\nsys.path += [\"/opt/conda/envs/rapids/lib\"]\n!cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/\n\nimport cupy as cp\nimport cusignal","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:41:19.600306Z","iopub.execute_input":"2024-03-30T05:41:19.601017Z","iopub.status.idle":"2024-03-30T05:42:49.500129Z","shell.execute_reply.started":"2024-03-30T05:41:19.600981Z","shell.execute_reply":"2024-03-30T05:42:49.499088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Environment functions","metadata":{}},{"cell_type":"code","source":"# Set the visible CUDA devices\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1\"\n\n# Set the strategy for using GPUs\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus) <= 1:\n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse:\n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n\n# Configure memory growth\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n    except RuntimeError as e:\n        print(e)\n\n# Enable or disable mixed precision\nMIX = True\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:49.501385Z","iopub.execute_input":"2024-03-30T05:42:49.501695Z","iopub.status.idle":"2024-03-30T05:42:50.708556Z","shell.execute_reply.started":"2024-03-30T05:42:49.501664Z","shell.execute_reply":"2024-03-30T05:42:50.707535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to set random seed for reproducibility\ndef set_random_seed(seed: int = 42, deterministic: bool = False):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    tf.random.set_seed(seed)\n    if deterministic:\n        os.environ['TF_DETERMINISTIC_OPS'] = '1'\n    else:\n        os.environ.pop('TF_DETERMINISTIC_OPS', None)\n\n# Set a deterministic behavior\nset_random_seed(deterministic=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.709847Z","iopub.execute_input":"2024-03-30T05:42:50.710136Z","iopub.status.idle":"2024-03-30T05:42:50.885302Z","shell.execute_reply.started":"2024-03-30T05:42:50.710112Z","shell.execute_reply":"2024-03-30T05:42:50.884558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data","metadata":{}},{"cell_type":"code","source":"def create_train_data():\n    # Read the dataset\n    df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n    \n    # Create a new identifier combining multiple columns\n    id_cols = ['eeg_id', 'spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n    df['new_id'] = df[id_cols].astype(str).agg('_'.join, axis=1)\n    \n    # Calculate the sum of votes for each class\n    df['sum_votes'] = df[CLASSES].sum(axis=1)\n    \n    # Group the data by the new identifier and aggregate various features\n    agg_functions = {\n        'eeg_id': 'first',\n        'eeg_label_offset_seconds': ['min', 'max'],\n        'spectrogram_label_offset_seconds': ['min', 'max'],\n        'spectrogram_id': 'first',\n        'patient_id': 'first',\n        'expert_consensus': 'first',\n        **{col: 'sum' for col in CLASSES},\n        'sum_votes': 'sum',\n    }\n    grouped_df = df.groupby('new_id').agg(agg_functions).reset_index()\n\n    # Flatten the MultiIndex columns and adjust column names\n    grouped_df.columns = [f\"{col[0]}_{col[1]}\" if col[1] else col[0] for col in grouped_df.columns]\n    grouped_df.columns = grouped_df.columns.str.replace('_first', '').str.replace('_sum', '')\n    \n    # Normalize the class columns\n    y_data = grouped_df[CLASSES].values\n    y_data_normalized = y_data / y_data.sum(axis=1, keepdims=True)\n    grouped_df[CLASSES] = y_data_normalized\n\n    # Split the dataset into high and low quality based on the sum of votes\n    high_quality_df = grouped_df[grouped_df['sum_votes'] >= 10].reset_index(drop=True)\n    low_quality_df = grouped_df[(grouped_df['sum_votes'] < 10) & (grouped_df['sum_votes'] >= 0)].reset_index(drop=True)\n\n    return high_quality_df, low_quality_df","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.886439Z","iopub.execute_input":"2024-03-30T05:42:50.886788Z","iopub.status.idle":"2024-03-30T05:42:50.897080Z","shell.execute_reply.started":"2024-03-30T05:42:50.886763Z","shell.execute_reply":"2024-03-30T05:42:50.896241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence):\n\n    def __init__(self, data, batch_size=32, shuffle=False, mode='train'):\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.mode = mode\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch.\"\"\"\n        return int(np.ceil(len(self.data) / self.batch_size))\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data.\"\"\"\n        indexes = self.indexes[index * self.batch_size : (index + 1) * self.batch_size]\n        X, y = self.__data_generation(indexes)\n        return X, y\n\n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch.\"\"\"\n        self.indexes = np.arange(len(self.data))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, indexes):\n        \"\"\"Generates data containing batch_size samples.\"\"\"\n        # Initialization\n        X = np.zeros((len(indexes), *SPEC_SIZE), dtype='float32')\n        y = np.zeros((len(indexes), len(CLASSES)), dtype='float32')\n\n        # Generate data\n        for j, i in enumerate(indexes):\n            row = self.data.iloc[i]\n            eeg_id = row['eeg_id']\n            spec_offset = int(row['spectrogram_label_offset_seconds_min'])\n            eeg_offset = int(row['eeg_label_offset_seconds_min'])\n            file_path = f'/kaggle/input/3-diff-time-specs-hms/images/{eeg_id}_{spec_offset}_{eeg_offset}.npz'\n            data = np.load(file_path)\n            eeg_data = data['final_image']\n            eeg_data_expanded = np.repeat(eeg_data[:, :, np.newaxis], 3, axis=2)\n\n            X[j] = eeg_data_expanded\n            if self.mode != 'test':\n                y[j] = row[CLASSES]\n\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.898131Z","iopub.execute_input":"2024-03-30T05:42:50.898712Z","iopub.status.idle":"2024-03-30T05:42:50.913015Z","shell.execute_reply.started":"2024-03-30T05:42:50.898683Z","shell.execute_reply":"2024-03-30T05:42:50.912199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"def lrfn(epoch):\n    lr_schedule = [1e-3, 1e-3, 1e-3, 1e-4, 1e-4, 1e-4, 1e-5, 1e-5, 1e-5]\n    return lr_schedule[epoch]\n\n# Define the learning rate scheduler callback\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)\n\nimport efficientnet.tfkeras as efn\n\ndef build_model():\n\n    # inp = tf.keras.Input(shape=(128,256,8))\n    inp = tf.keras.Input(shape=(1536, 512, 3))\n\n    base_model = efn.EfficientNetB0(include_top=False, weights=None, input_shape=None)\n    base_model.load_weights('/kaggle/input/tf-efficientnet-imagenet-weights/efficientnet-b0_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5')\n\n    # # RESHAPE INPUT 128x256x8 => 512x512x3 MONOTONE IMAGE\n    # # KAGGLE SPECTROGRAMS\n    # x1 = [inp[:,:,:,i:i+1] for i in range(4)]\n    # x1 = tf.keras.layers.Concatenate(axis=1)(x1)\n    # # EEG SPECTROGRAMS\n    # x2 = [inp[:,:,:,i+4:i+5] for i in range(4)]\n    # x2 = tf.keras.layers.Concatenate(axis=1)(x2)\n    # # MAKE 512X512X3\n    # if USE_KAGGLE_SPECTROGRAMS & USE_EEG_SPECTROGRAMS:\n    #     x = tf.keras.layers.Concatenate(axis=2)([x1,x2])\n    # elif USE_EEG_SPECTROGRAMS: x = x2\n    # else: x = x1\n    # x = tf.keras.layers.Concatenate(axis=3)([x,x,x])\n\n    # OUTPUT\n    x = base_model(inp)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dense(6,activation='softmax', dtype='float32')(x)\n\n    # COMPILE MODEL\n    model = tf.keras.Model(inputs=inp, outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate = 1e-3)\n    loss = tf.keras.losses.KLDivergence()\n\n    model.compile(loss=loss, optimizer = opt)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.914289Z","iopub.execute_input":"2024-03-30T05:42:50.914623Z","iopub.status.idle":"2024-03-30T05:42:50.930630Z","shell.execute_reply.started":"2024-03-30T05:42:50.914595Z","shell.execute_reply":"2024-03-30T05:42:50.929981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## cross_validate_model - Label Refine","metadata":{}},{"cell_type":"code","source":"def cross_validate_model(train_data, train_data_2, folds, random_seed, targets, nome_modelo):\n    inicio = time.time()\n    path_model = f'MLP_Model{nome_modelo}'\n    if not os.path.exists(path_model):\n        os.makedirs(path_model)\n\n    all_oof = []\n    all_oof2 = []\n    all_true = []\n    models = []\n    score_list = []\n    \n    # Separating the data to iterate over both dataframes simultaneously\n    gkf = StratifiedGroupKFold(n_splits=folds, shuffle=True, random_state=random_seed)\n    splits1 = list(gkf.split(train_data, train_data[[\"expert_consensus\"]], train_data[\"patient_id\"]))\n    splits2 = list(gkf.split(train_data_2, train_data_2[[\"expert_consensus\"]], train_data_2[\"patient_id\"]))\n\n    # Iterate over folds in parallel\n    for i, ((train_index, valid_index), (train_index2, valid_index2)) in enumerate(zip(splits1, splits2)):\n        \n        # Copy the dataframes to avoid leaks\n        train_data_ = train_data.copy()\n        train_data_2_ = train_data_2.copy()\n        set_random_seed(random_seed, deterministic=True)\n        \n        # Start folding\n        print('#' * 25)\n        print(f'### Fold {i + 1}')\n        print(f'### train size 1 {len(train_index)}, valid size {len(valid_index)}')\n        print(f'### train size 2 {len(train_index2)}, valid size {len(valid_index2)}')\n        print('#' * 25)\n\n        ### --------------------------- Performs model 1 training -------------- --------------------------- ###\n        K.clear_session()\n        train_gen = DataGenerator(train_data_.iloc[train_index], shuffle=True, batch_size=BATCH)\n        valid_gen = DataGenerator(train_data_.iloc[valid_index], shuffle=False, batch_size=(BATCH*2), mode='valid')\n        model = build_EfficientNetB0(input_shape=(512, 512, 3), num_classes=6)\n        history = model.fit(train_gen, verbose=2, validation_data=valid_gen, epochs=EPOCHS, callbacks=[LR])\n\n        # Model training result 1\n        train_loss = history.history['loss'][-1]  \n        valid_loss = history.history['val_loss'][-1]\n        print(f'train_loss 1 {train_loss} valid_loss 1 {valid_loss}')\n        score_list.append((train_loss, valid_loss))\n\n        \n        ### --------------------------- creation of pseudo labels ---------------- ------------------------- ###\n        # pseudo labels for low quality data\n        train_2_index_total_gen = DataGenerator(train_data_2_.iloc[train_index2], shuffle=False, batch_size=BATCH)\n        pseudo_labels_2 = model.predict(train_2_index_total_gen, verbose=2)\n        # Refinement of low quality labels\n        train_data_2_.loc[train_index2, TARGETS] /= 2\n        train_data_2_.loc[train_index2, TARGETS] += pseudo_labels_2 / 2\n\n        # pseudo labels for high quality data (50% of data)\n        train_data_3_ = train_data_\n        train_3_index_total_gen = DataGenerator(train_data_3_.iloc[train_index], shuffle=False, batch_size=BATCH)\n        pseudo_labels_3 = model.predict(train_3_index_total_gen, verbose=2)\n        # Refinement of high quality labels\n        train_data_3_.loc[train_index, TARGETS] /= 2\n        train_data_3_.loc[train_index, TARGETS] += pseudo_labels_3 / 2\n\n        ### --------------------------- Creation of the data generator for the refined labels model --------- -------------------------------- ###\n        # Low quality data\n        np.random.shuffle(train_index)\n        np.random.shuffle(valid_index)\n        sixty_percent_length = int(0.5 * len(train_data_3_))\n        train_index_60 = train_index[:int(sixty_percent_length * len(train_index) / len(train_data_3_))]\n        valid_index_60 = valid_index[:int(sixty_percent_length * len(valid_index) / len(train_data_3_))]\n        train_gen_2 = DataGenerator(pd.concat([train_data_3_.iloc[train_index_60], train_data_2_.iloc[train_index2]]), shuffle=True, batch_size=BATCH)\n        valid_gen_2 = DataGenerator(pd.concat([train_data_3_.iloc[valid_index_60], train_data_2_.iloc[valid_index2]]), shuffle=False, batch_size=BATCH*2, mode='valid')\n        # Rebuild the high quality data generator with 50% of the labels refined\n        train_gen = DataGenerator(train_data_.iloc[train_index], shuffle=True, batch_size=BATCH)\n        valid_gen = DataGenerator(train_data_.iloc[valid_index], shuffle=False, batch_size=(BATCH*2), mode='valid')\n        \n        ### --------------------------- Model 2 training and finetunning -------------- --------------------------- ###\n        K.clear_session()\n        new_model = build_EfficientNetB0(input_shape=(512, 512, 3), num_classes=6)\n        # Training with the refined low-quality data\n        history = new_model.fit(train_gen_2, verbose=2, validation_data=valid_gen_2, epochs=EPOCHS, callbacks=[LR])\n        # Finetuning with refined high-quality data\n        history = new_model.fit(train_gen, verbose=2, validation_data=valid_gen, epochs=EPOCHS, callbacks=[LR])\n        new_model.save_weights(f'{path_model}/MLP_fold{i}.weights.h5')\n        models.append(new_model)\n\n        # Model 2 training result\n        train_loss = history.history['loss'][-1]  # Valor da perda do último epoch de treinamento\n        valid_loss = history.history['val_loss'][-1]  # Valor da perda do último epoch de validação\n        print(f'train_loss 2 {train_loss} valid_loss 2 {valid_loss}')\n        score_list.append((train_loss, valid_loss))\n\n\n        # MLP OOF\n        oof = new_model.predict(valid_gen, verbose=2)\n        all_oof.append(oof)\n        all_true.append(train_data.iloc[valid_index][TARGETS].values)\n\n        # TRAIN MEAN OOF\n        y_train = train_data.iloc[train_index][targets].values\n        y_valid = train_data.iloc[valid_index][targets].values\n        oof = y_valid.copy()\n        for j in range(6):\n            oof[:,j] = y_train[:,j].mean()\n        oof = oof / oof.sum(axis=1,keepdims=True)\n        all_oof2.append(oof)\n\n        del model, new_model, train_gen, valid_gen, train_2_index_total_gen, train_gen_2, valid_gen_2, oof, y_train, y_valid, train_index, valid_index\n        K.clear_session()\n        gc.collect()\n\n        if i==folds-1: break\n\n    all_oof = np.concatenate(all_oof)\n    all_oof2 = np.concatenate(all_oof2)\n    all_true = np.concatenate(all_true)\n\n    oof = pd.DataFrame(all_oof.copy())\n    oof['id'] = np.arange(len(oof))\n\n    true = pd.DataFrame(all_true.copy())\n    true['id'] = np.arange(len(true))\n\n    cv = score(solution=true, submission=oof, row_id_column_name='id')\n    fim = time.time()\n    tempo_execucao = fim - inicio\n    print(f'{nome_modelo} CV Score with EEG Spectrograms ={cv} tempo: {tempo_execucao}')\n    \n    gc.collect()\n\n    score_array = np.array(score_list)\n    std_dev = np.std(score_array, axis=0)\n    std_dev = std_dev.tolist()\n\n    return cv, tempo_execucao, all_oof, all_oof2, all_true, models, score_list, std_dev, path_model","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.933931Z","iopub.execute_input":"2024-03-30T05:42:50.934230Z","iopub.status.idle":"2024-03-30T05:42:50.962383Z","shell.execute_reply.started":"2024-03-30T05:42:50.934209Z","shell.execute_reply":"2024-03-30T05:42:50.961503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not ONLY_INFERENCE:\n    high_quality_df, low_quality_df = create_train_data()\n    result, tempo_execucao, all_oof, all_oof2, all_true, models, score_list, std_dev, path_model = cross_validate_model(high_quality_df, low_quality_df, FOLDS, 42, CLASSES, NAME)\n    print(f'Result cv V1 final {result}{tempo_execucao} {score_list} {std_dev}')\n    display(result)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.963431Z","iopub.execute_input":"2024-03-30T05:42:50.963772Z","iopub.status.idle":"2024-03-30T05:42:50.979573Z","shell.execute_reply.started":"2024-03-30T05:42:50.963743Z","shell.execute_reply":"2024-03-30T05:42:50.978891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# INFERENCE","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:50.980637Z","iopub.execute_input":"2024-03-30T05:42:50.981172Z","iopub.status.idle":"2024-03-30T05:42:51.019196Z","shell.execute_reply.started":"2024-03-30T05:42:50.981142Z","shell.execute_reply":"2024-03-30T05:42:51.018381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ ALL SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\nfiles2 = os.listdir(PATH2)\nprint(f'There are {len(files2)} test spectrogram parquets')\n    \nspectrograms2 = {}\nfor i,f in enumerate(files2):\n    if i%100==0: print(i,', ',end='')\n    tmp = pd.read_parquet(f'{PATH2}{f}')\n    name = int(f.split('.')[0])\n    spectrograms2[name] = tmp.iloc[:,1:].values\n    \n# RENAME FOR DATALOADER\ntest = test.rename({'spectrogram_id':'spec_id'},axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:51.020160Z","iopub.execute_input":"2024-03-30T05:42:51.020406Z","iopub.status.idle":"2024-03-30T05:42:51.359993Z","shell.execute_reply.started":"2024-03-30T05:42:51.020385Z","shell.execute_reply":"2024-03-30T05:42:51.359014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt, librosa\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:51.364186Z","iopub.execute_input":"2024-03-30T05:42:51.364589Z","iopub.status.idle":"2024-03-30T05:42:51.511528Z","shell.execute_reply.started":"2024-03-30T05:42:51.364549Z","shell.execute_reply":"2024-03-30T05:42:51.510739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ ALL EEG SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 1\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:42:51.512468Z","iopub.execute_input":"2024-03-30T05:42:51.513046Z","iopub.status.idle":"2024-03-30T05:43:07.204429Z","shell.execute_reply.started":"2024-03-30T05:42:51.513020Z","shell.execute_reply":"2024-03-30T05:43:07.203486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_spectrogram_with_cusignal(eeg_data, eeg_id, start, duration= 50,\n                                    low_cut_freq = 0.7, high_cut_freq = 20, order_band = 5,\n                                    spec_size_freq = 267, spec_size_time = 30,\n                                    nperseg_ = 1500, noverlap_ = 1483, nfft_ = 2750,\n                                    sigma_gaussian = 0.7, \n                                    mean_montage_names = 4):\n    \n    electrode_names = ['LL', 'RL', 'LP', 'RP']\n\n    electrode_pairs = [\n        ['Fp1', 'F7', 'T3', 'T5', 'O1'],\n        ['Fp2', 'F8', 'T4', 'T6', 'O2'],\n        ['Fp1', 'F3', 'C3', 'P3', 'O1'],\n        ['Fp2', 'F4', 'C4', 'P4', 'O2']\n    ]\n    \n    # Filter specifications\n    nyquist_freq = 0.5 * 200\n    low_cut_freq_normalized = low_cut_freq / nyquist_freq\n    high_cut_freq_normalized = high_cut_freq / nyquist_freq\n\n    # Bandpass and notch filter\n    bandpass_coefficients = butter(order_band, [low_cut_freq_normalized, high_cut_freq_normalized], btype='band')\n    notch_coefficients = iirnotch(w0=60, Q=30, fs=200)\n    \n    spec_size = duration * 200\n    start = start * 200\n    real_start = start + (10_000//2) - (spec_size//2)\n    eeg_data = eeg_data.iloc[real_start:real_start+spec_size]\n    \n    \n    # Spectrogram parameters\n    fs = 200\n    nperseg = nperseg_\n    noverlap = noverlap_\n    nfft = nfft_\n    \n    if spec_size_freq <=0 or spec_size_time <=0:\n        frequencias_size = int((nfft // 2)/5.15198)+1\n        segmentos = int((spec_size - noverlap) / (nperseg - noverlap)) \n    else:\n        frequencias_size = spec_size_freq\n        segmentos = spec_size_time\n        \n    spectrogram = cp.zeros((frequencias_size, segmentos, 4), dtype='float32')\n    \n    processed_eeg = {}\n\n    for i, name in enumerate(electrode_names):\n        cols = electrode_pairs[i]\n        processed_eeg[name] = np.zeros(spec_size)\n        for j in range(4):\n            # Compute differential signals\n            signal = cp.array(eeg_data[cols[j]].values - eeg_data[cols[j+1]].values)\n\n            # Handle NaNs\n            mean_signal = cp.nanmean(signal)\n            signal = cp.nan_to_num(signal, nan=mean_signal) if cp.isnan(signal).mean() < 1 else cp.zeros_like(signal)\n            \n\n            # Filter bandpass and notch\n            signal_filtered = filtfilt(*notch_coefficients, signal.get())\n            signal_filtered = filtfilt(*bandpass_coefficients, signal_filtered)\n            signal = cp.asarray(signal_filtered)\n            \n            frequencies, times, Sxx = cusignal.spectrogram(signal, fs, nperseg=nperseg, noverlap=noverlap, nfft=nfft)\n\n            # Filter frequency range\n            valid_freqs = (frequencies >= 0.59) & (frequencies <= 20)\n            frequencies_filtered = frequencies[valid_freqs]\n            Sxx_filtered = Sxx[valid_freqs, :]\n\n            # Logarithmic transformation and normalization using Cupy\n            spectrogram_slice = cp.clip(Sxx_filtered, cp.exp(-4), cp.exp(6))\n            spectrogram_slice = cp.log10(spectrogram_slice)\n\n            normalization_epsilon = 1e-6\n            mean = spectrogram_slice.mean(axis=(0, 1), keepdims=True)\n            std = spectrogram_slice.std(axis=(0, 1), keepdims=True)\n            spectrogram_slice = (spectrogram_slice - mean) / (std + normalization_epsilon)\n            \n            spectrogram[:, :, i] += spectrogram_slice\n            processed_eeg[f'{cols[j]}_{cols[j+1]}'] = signal.get()\n            processed_eeg[name] += signal.get()\n        \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        if mean_montage_names > 0:\n            spectrogram[:,:,i] /= mean_montage_names\n\n    # Convert to NumPy and apply Gaussian filter\n    spectrogram_np = cp.asnumpy(spectrogram)\n    if sigma_gaussian > 0.0:\n        spectrogram_np = gaussian_filter(spectrogram_np, sigma=sigma_gaussian)\n\n    # Filter EKG signal\n    ekg_signal_filtered = filtfilt(*notch_coefficients, eeg_data[\"EKG\"].values)\n    ekg_signal_filtered = filtfilt(*bandpass_coefficients, ekg_signal_filtered)\n    processed_eeg['EKG'] = np.array(ekg_signal_filtered)\n\n    return spectrogram_np, processed_eeg","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:43:07.205873Z","iopub.execute_input":"2024-03-30T05:43:07.206415Z","iopub.status.idle":"2024-03-30T05:43:07.226208Z","shell.execute_reply.started":"2024-03-30T05:43:07.206390Z","shell.execute_reply":"2024-03-30T05:43:07.225303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_spectogram_competition(spec_id, seconds_min):\n    spec = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/{spec_id}.parquet')\n    inicio = (seconds_min) // 2\n    img = spec.fillna(0).values[:, 1:].T.astype(\"float32\")\n    img = img[:, inicio:inicio+300]\n    \n    # Log transform and normalize\n    img = np.clip(img, np.exp(-4), np.exp(6))\n    img = np.log(img)\n    eps = 1e-6\n    img_mean = img.mean()\n    img_std = img.std()\n    img = (img - img_mean) / (img_std + eps)\n    \n    return img ","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:43:07.227586Z","iopub.execute_input":"2024-03-30T05:43:07.227928Z","iopub.status.idle":"2024-03-30T05:43:07.251020Z","shell.execute_reply.started":"2024-03-30T05:43:07.227896Z","shell.execute_reply":"2024-03-30T05:43:07.250256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom tqdm import tqdm\nimport pandas as pd\nimport cv2\nimport os\nimport matplotlib.pyplot as plt\nall_eegs2 = {}\n# Make sure the 'images' folder exists\noutput_folder = 'imagens'\nif not os.path.exists(output_folder):\n    os.makedirs(output_folder)\n    \ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape:',test.shape)\nprint(test.head())\n\n# Creation of spectograms on the test base\nfor i in tqdm(range(len(test)), desc=\"Processing EEGs\"):\n    row = test.iloc[i]\n    eeg_id = row['eeg_id']\n    spec_id = row['spectrogram_id']\n    seconds_min = 0\n    start_second = 0\n    eeg_data = pd.read_parquet(f'/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/{eeg_id}.parquet')\n    eeg_new_key = eeg_id\n    image_50s, _ = create_spectrogram_with_cusignal(eeg_data=eeg_data, eeg_id=eeg_id, start=start_second, duration= 50,\n                                    low_cut_freq = 0.7, high_cut_freq = 20, order_band = 5,\n                                    spec_size_freq = 267, spec_size_time = 501,\n                                    nperseg_ = 1500, noverlap_ = 1483, nfft_ = 2750,\n                                    sigma_gaussian = 0.0, \n                                    mean_montage_names = 4)\n    image_10s, _ = create_spectrogram_with_cusignal(eeg_data=eeg_data, eeg_id=eeg_id, start=start_second, duration= 10,\n                                    low_cut_freq = 0.7, high_cut_freq = 20, order_band = 5,\n                                    spec_size_freq = 100, spec_size_time = 291,\n                                    nperseg_ = 260, noverlap_ = 254, nfft_ = 1030,\n                                    sigma_gaussian = 0.0, \n                                    mean_montage_names = 4)\n    image_10m = create_spectogram_competition(spec_id, seconds_min)\n    \n    imagem_final_unico_canal = np.zeros((1068, 501))\n    for j in range(4):\n        inicio = j * 267 \n        fim = inicio + 267\n        imagem_final_unico_canal[inicio:fim, :] = image_50s[:, :, j]\n        \n    \n    imagem_final_unico_canal2 = np.zeros((400, 291))\n    for n in range(4):\n        inicio = n * 100 \n        fim = inicio + 100\n        imagem_final_unico_canal2[inicio:fim, :] = image_10s[:, :, n]\n    \n    imagem_final_unico_canal_resized = cv2.resize(imagem_final_unico_canal, (400, 800), interpolation=cv2.INTER_AREA)\n    imagem_final_unico_canal2_resized = cv2.resize(imagem_final_unico_canal2, (300, 400), interpolation=cv2.INTER_AREA)\n    eeg_new_resized = cv2.resize(image_10m, (300, 400), interpolation=cv2.INTER_AREA)\n    imagem_final = np.zeros((800, 700), dtype=np.float32)\n    imagem_final[0:800, 0:400] = imagem_final_unico_canal_resized\n    imagem_final[0:400,400:700] = imagem_final_unico_canal2_resized\n    imagem_final[400:800, 400:700] = eeg_new_resized\n    imagem_final = imagem_final[::-1]\n    \n    imagem_final = cv2.resize(imagem_final, (512, 512), interpolation=cv2.INTER_AREA)\n    \n    all_eegs2[eeg_new_key] = imagem_final\n    \n    if i ==0:\n        plt.figure(figsize=(10, 10))\n        plt.imshow(imagem_final, cmap='jet')\n        plt.axis('off')\n        plt.show()\n\n        print(imagem_final.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:43:07.252178Z","iopub.execute_input":"2024-03-30T05:43:07.252454Z","iopub.status.idle":"2024-03-30T05:44:54.095283Z","shell.execute_reply.started":"2024-03-30T05:43:07.252430Z","shell.execute_reply":"2024-03-30T05:44:54.094354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nclass DataGeneratorTest(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, specs, eeg_specs, batch_size=32, shuffle=False, eegs={}, mode='train'):\n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.eegs = eegs\n        self.mode = mode\n        self.specs = specs\n        self.eeg_specs = eeg_specs\n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int( np.ceil( len(self.data) / self.batch_size ) )\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X, y = self.__data_generation(indexes)\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.data) )\n        if self.shuffle: np.random.shuffle(self.indexes)\n\n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples'\n        X = np.zeros((len(indexes),128,256,8),dtype='float32')\n        y = np.zeros((len(indexes),6),dtype='float32')\n        img = np.ones((128,256),dtype='float32')\n\n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]\n            if self.mode=='test':\n                r = 0\n            else:\n                r = int( (row['spectrogram_label_offset_seconds_min'] + row['spectrogram_label_offset_seconds_max'])//4 )\n\n            for k in range(4):\n                # EXTRACT 300 ROWS OF SPECTROGRAM\n                img = self.specs[row.spectrogram_id][r:r+300,k*100:(k+1)*100].T\n\n                # LOG TRANSFORM SPECTROGRAM\n                img = np.clip(img,np.exp(-4),np.exp(8))\n                img = np.log(img)\n\n                # STANDARDIZE PER IMAGE\n                ep = 1e-6\n                m = np.nanmean(img.flatten())\n                s = np.nanstd(img.flatten())\n                img = (img-m)/(s+ep)\n                img = np.nan_to_num(img, nan=0.0)\n\n                # CROP TO 256 TIME STEPS\n                X[j,14:-14,:,k] = img[:,22:-22] / 2.0\n\n            # EEG SPECTROGRAMS\n            img = self.eeg_specs[row.eeg_id]\n            X[j,:,:,4:] = img\n\n            if self.mode!='test':\n                y[j,] = row[TARGETS]\n        \n        # EEG信号\n        x1 = [X[:,:,:,i:i+1] for i in range(4)]\n        x1 = np.concatenate(x1, axis=1)\n\n        # EEG频谱\n        x2 = [X[:,:,:,i+4:i+5] for i in range(4)]\n        x2 = np.concatenate(x2, axis=1)\n\n        X = np.concatenate([x1, x2], axis=2)\n        \n#         X = np.zeros((len(indexes),SPEC_SIZE[0],SPEC_SIZE[1],SPEC_SIZE[2]),dtype='float32')\n        X2 = np.zeros((len(indexes),SPEC_SIZE[0],SPEC_SIZE[1],1),dtype='float32')\n\n        y = np.zeros((len(indexes),6),dtype='float32')\n\n        for j,i in enumerate(indexes):\n            row = self.data.iloc[i]\n            eeg_data = self.eegs[row.eeg_id] \n            eeg_data_expanded = eeg_data[:, :, np.newaxis]\n#             eeg_data_expanded = np.repeat(eeg_data[:, :, np.newaxis], 3, axis=2)\n            X2[j,] = eeg_data_expanded\n            if self.mode!='test':\n                y[j] = row[CLASSES]\n                \n        X = np.concatenate([X, X2, X2], axis=1)\n        X = np.repeat(X, 3, axis=3)\n        \n        return X,y","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:44:54.096622Z","iopub.execute_input":"2024-03-30T05:44:54.096915Z","iopub.status.idle":"2024-03-30T05:44:54.118069Z","shell.execute_reply.started":"2024-03-30T05:44:54.096891Z","shell.execute_reply":"2024-03-30T05:44:54.117028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER MLP ON TEST\npreds = []\nmodel = build_model()\ntest_gen = DataGeneratorTest(test, specs=spectrograms2, eeg_specs=all_eegs, shuffle=False, batch_size=BATCH, eegs=all_eegs2, mode='test')\n\nprint('Inferring test... ',end='')\nfor i in range(FOLDS):\n    print(f'fold {i+1}, ',end='')\n    model.load_weights(f'/kaggle/input/mactf-v2/TF3/EffNet_v5_f{i}.h5')\n    pred = model.predict(test_gen, verbose=0)\n    preds.append(pred)\npred = np.mean(preds,axis=0)\nprint()\nprint('Test preds shape',pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:44:54.119505Z","iopub.execute_input":"2024-03-30T05:44:54.120075Z","iopub.status.idle":"2024-03-30T05:45:04.280306Z","shell.execute_reply.started":"2024-03-30T05:44:54.120042Z","shell.execute_reply":"2024-03-30T05:45:04.279396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATE SUBMISSION.CSV\nfrom IPython.display import display\n\nsub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\ndisplay( sub.head() )","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:45:04.281640Z","iopub.execute_input":"2024-03-30T05:45:04.282013Z","iopub.status.idle":"2024-03-30T05:45:04.302704Z","shell.execute_reply.started":"2024-03-30T05:45:04.281979Z","shell.execute_reply":"2024-03-30T05:45:04.301561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nprint(sub.iloc[:,-6:].sum(axis=1))","metadata":{"execution":{"iopub.status.busy":"2024-03-30T05:45:04.304116Z","iopub.execute_input":"2024-03-30T05:45:04.304417Z","iopub.status.idle":"2024-03-30T05:45:04.311319Z","shell.execute_reply.started":"2024-03-30T05:45:04.304392Z","shell.execute_reply":"2024-03-30T05:45:04.310427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}