{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nprint('TensorFlow version =',tf.__version__)\n\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-14T17:49:13.399585Z","iopub.execute_input":"2024-03-14T17:49:13.400309Z","iopub.status.idle":"2024-03-14T17:49:25.021423Z","shell.execute_reply.started":"2024-03-14T17:49:13.400276Z","shell.execute_reply":"2024-03-14T17:49:25.020325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\n\n# USE MULTIPLE GPUS\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n\nVER = 5\n\n\nLOAD_MODELS_FROM = '/kaggle/input/brain-efficientnet-models-v3-v4-v5/'\n\nUSE_KAGGLE_SPECTROGRAMS = True\nUSE_EEG_SPECTROGRAMS = True","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:49:25.023101Z","iopub.execute_input":"2024-03-14T17:49:25.023426Z","iopub.status.idle":"2024-03-14T17:49:25.300058Z","shell.execute_reply.started":"2024-03-14T17:49:25.023399Z","shell.execute_reply":"2024-03-14T17:49:25.298688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:49:25.301461Z","iopub.execute_input":"2024-03-14T17:49:25.301868Z","iopub.status.idle":"2024-03-14T17:49:25.399714Z","shell.execute_reply.started":"2024-03-14T17:49:25.301828Z","shell.execute_reply":"2024-03-14T17:49:25.398827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:49:25.400915Z","iopub.execute_input":"2024-03-14T17:49:25.401275Z","iopub.status.idle":"2024-03-14T17:50:32.297128Z","shell.execute_reply.started":"2024-03-14T17:49:25.401243Z","shell.execute_reply":"2024-03-14T17:50:32.296057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:32.300552Z","iopub.execute_input":"2024-03-14T17:50:32.300861Z","iopub.status.idle":"2024-03-14T17:50:39.324193Z","shell.execute_reply.started":"2024-03-14T17:50:32.300838Z","shell.execute_reply":"2024-03-14T17:50:39.323068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install albumentations\n\nimport albumentations as albu\n\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, batch_size=32, shuffle=False, augment=False, mode='train', specs=spectrograms): \n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.mode = mode\n        self.specs = specs\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int(np.ceil(len(self.data) / self.batch_size))\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(indexes)\n        if self.augment:\n            X = self.__augment_batch(X) \n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.data))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' \n        \n        X = np.zeros((len(indexes), 128, 256, 4), dtype='float32')\n        y = np.zeros((len(indexes), 6), dtype='float32')\n        \n        for j, i in enumerate(indexes):\n            row = self.data.iloc[i]\n            if self.mode == 'test': \n                r = 0\n            else: \n                r = int((row['min'] + row['max']) // 4)\n\n            for k in range(4):\n                # EXTRACT 300 ROWS OF SPECTROGRAM\n                img = self.specs[row.spec_id][r:r+300, k*100:(k+1)*100].T\n                \n                # LOG TRANSFORM SPECTROGRAM\n                img = np.clip(img, np.exp(-4), np.exp(8))\n                img = np.log(img)\n                \n                # STANDARDIZE PER IMAGE\n                ep = 1e-6\n                m = np.nanmean(img.flatten())\n                s = np.nanstd(img.flatten())\n                img = (img - m) / (s + ep)\n                img = np.nan_to_num(img, nan=0.0)\n                \n                # CROP TO 256 TIME STEPS\n                X[j, 14:-14, :, k] = img[:, 22:-22] / 2.0\n        \n            if self.mode != 'test':\n                y[j] = row[TARGETS]\n            \n        return X, y\n    \n    def __random_transform(self, img):\n        composition = albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            # Add more transformations here if needed\n        ])\n        return composition(image=img)['image']\n            \n    def __augment_batch(self, img_batch):\n        for i in range(img_batch.shape[0]):\n            img_batch[i] = self.__random_transform(img_batch[i])\n        return img_batch\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:39.325395Z","iopub.execute_input":"2024-03-14T17:50:39.325748Z","iopub.status.idle":"2024-03-14T17:50:53.832886Z","shell.execute_reply.started":"2024-03-14T17:50:39.325714Z","shell.execute_reply":"2024-03-14T17:50:53.832063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gen = DataGenerator(train, batch_size=32, shuffle=False)\n\nROWS = 2\nCOLS = 2  # Solo usaremos 4 espectrogramas ahora\nBATCHES = 2\n\nfor i, (x, y) in enumerate(gen):\n    plt.figure(figsize=(20, 8))\n    for j in range(ROWS):\n        for k in range(COLS):\n            plt.subplot(ROWS, COLS, j * COLS + k + 1)\n            t = y[j * COLS + k]\n            img = x[j * COLS + k, :, :, 0][::-1, ]  # Seleccionamos solo el primer canal del espectrograma\n            mn = img.flatten().min()\n            mx = img.flatten().max()\n            img = (img - mn) / (mx - mn)  # Normalizamos la imagen\n            plt.imshow(img)\n            tars = f'[{t[0]:0.2f}'\n            for s in t[1:]:\n                tars += f', {s:0.2f}'\n            eeg = train.eeg_id.values[i * 32 + j * COLS + k]\n            plt.title(f'EEG = {eeg}\\nTarget = {tars}', size=12)\n            plt.yticks([])\n            plt.ylabel('Frequencies (Hz)', size=14)\n            plt.xlabel('Time (sec)', size=16)\n    plt.show()\n    if i == BATCHES - 1:\n        break\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:53.834073Z","iopub.execute_input":"2024-03-14T17:50:53.834588Z","iopub.status.idle":"2024-03-14T17:50:55.850521Z","shell.execute_reply.started":"2024-03-14T17:50:53.834563Z","shell.execute_reply":"2024-03-14T17:50:55.849631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key, value in spectrograms.items():\n    print(f\"Forma de {key}: {value.shape}\")","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-14T17:50:55.851986Z","iopub.execute_input":"2024-03-14T17:50:55.852845Z","iopub.status.idle":"2024-03-14T17:50:55.941834Z","shell.execute_reply.started":"2024-03-14T17:50:55.852807Z","shell.execute_reply":"2024-03-14T17:50:55.940906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    inp = tf.keras.Input(shape=(128,256,4))  # Ajustamos la forma de entrada para coincidir con el modelo de CNN\n\n    # RESHAPE INPUT 128x256x4 => 512x512x3 MONOTONE IMAGE\n    # KAGGLE SPECTROGRAMS\n    x1 = [inp[:,:,:,i:i+1] for i in range(4)]\n    x1 = tf.keras.layers.Concatenate(axis=1)(x1)\n    # EEG SPECTROGRAMS\n    x2 = [inp[:,:,:,i+4:i+5] for i in range(4)]\n    x2 = tf.keras.layers.Concatenate(axis=1)(x2)\n    # MAKE 512X512X3\n    if USE_KAGGLE_SPECTROGRAMS & USE_EEG_SPECTROGRAMS:\n        x = tf.keras.layers.Concatenate(axis=2)([x1,x2])\n    elif USE_EEG_SPECTROGRAMS: \n        x = x2\n    else: \n        x = x1\n    x = tf.keras.layers.Concatenate(axis=3)([x,x,x])","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:55.943457Z","iopub.execute_input":"2024-03-14T17:50:55.944068Z","iopub.status.idle":"2024-03-14T17:50:55.953117Z","shell.execute_reply.started":"2024-03-14T17:50:55.944033Z","shell.execute_reply":"2024-03-14T17:50:55.952078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Input\n\ndef build_model():\n    inp = Input(shape=(128, 256, 4))  # Definir la entrada del modelo\n    \n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(64, activation='relu'),\n        Dense(6, activation='softmax')\n    ])\n    \n    opt = tf.keras.optimizers.Adam(learning_rate=1e-3)\n    loss = tf.keras.losses.CategoricalCrossentropy()\n\n    model.compile(loss=loss, optimizer=opt, metrics=['accuracy'])\n    \n    return model\n\n# Construir y compilar el modelo\nmodel = build_model()\n\n# Resumen del modelo\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:55.954316Z","iopub.execute_input":"2024-03-14T17:50:55.954586Z","iopub.status.idle":"2024-03-14T17:50:56.337131Z","shell.execute_reply.started":"2024-03-14T17:50:55.954564Z","shell.execute_reply":"2024-03-14T17:50:56.336190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Construir el modelo\nmodel.build(input_shape=(None, 128, 256, 4))\n\n# Resumen del modelo\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:56.338176Z","iopub.execute_input":"2024-03-14T17:50:56.338458Z","iopub.status.idle":"2024-03-14T17:50:56.422864Z","shell.execute_reply.started":"2024-03-14T17:50:56.338435Z","shell.execute_reply":"2024-03-14T17:50:56.421896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy import interpolate\n\n# Define la función de interpolación para redimensionar los espectrogramas\ndef resize_spectrogram(spectrogram, target_shape):\n    x = np.arange(spectrogram.shape[1])\n    y = np.arange(spectrogram.shape[0])\n    interpolator = interpolate.interp2d(x, y, spectrogram, kind='linear')\n    x_new = np.linspace(0, spectrogram.shape[1] - 1, target_shape[1])\n    y_new = np.linspace(0, spectrogram.shape[0] - 1, target_shape[0])\n    resized_spectrogram = interpolator(x_new, y_new)\n    \n    # Rellenar valores faltantes después de la interpolación\n    resized_spectrogram = np.nan_to_num(resized_spectrogram, nan=0.0)\n    \n    return resized_spectrogram\n\n# Obtener las formas de todos los espectrogramas\nshapes = [spectrogram.shape for spectrogram in spectrograms.values()]\n\n# Obtener la forma máxima entre todas las formas\nmax_shape = np.max(shapes, axis=0)\nprint(\"Forma máxima:\", max_shape)\n\n# Redimensionar todos los espectrogramas para que tengan la misma forma que la forma máxima\nspectrograms_resized = {}\nfor id, spectrogram in spectrograms.items():\n    # Redimensionar el espectrograma para que tenga la misma forma que la forma máxima\n    resized_spectrogram = resize_spectrogram(spectrogram, max_shape)\n    spectrograms_resized[id] = resized_spectrogram\n\n# Convertir los espectrogramas redimensionados en un array de NumPy\nX = np.array(list(spectrograms_resized.values()))\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T17:50:56.424079Z","iopub.execute_input":"2024-03-14T17:50:56.424368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(list(spectrograms.values()))\ny = df[TARGETS].values\n\n# División de los datos en entrenamiento y prueba\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Entrenamiento del modelo\nhistory = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_split=0.2)\n\n# Evaluación del modelo\ntest_loss, test_acc = model.evaluate(X_test, y_test)\nprint('Test accuracy:', test_acc)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}