{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30665,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nprint('TensorFlow version =',tf.__version__)\n\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-21T18:21:16.463348Z","iopub.execute_input":"2024-03-21T18:21:16.464231Z","iopub.status.idle":"2024-03-21T18:21:30.675926Z","shell.execute_reply.started":"2024-03-21T18:21:16.464197Z","shell.execute_reply":"2024-03-21T18:21:30.674781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\n\n# USE MULTIPLE GPUS\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n\nVER = 5\n\n\nLOAD_MODELS_FROM = '/kaggle/input/brain-efficientnet-models-v3-v4-v5/'\n\nUSE_KAGGLE_SPECTROGRAMS = True\nUSE_EEG_SPECTROGRAMS = False","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:21:30.677711Z","iopub.execute_input":"2024-03-21T18:21:30.678023Z","iopub.status.idle":"2024-03-21T18:21:30.992676Z","shell.execute_reply.started":"2024-03-21T18:21:30.677996Z","shell.execute_reply":"2024-03-21T18:21:30.991673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:21:30.993903Z","iopub.execute_input":"2024-03-21T18:21:30.994213Z","iopub.status.idle":"2024-03-21T18:21:31.126672Z","shell.execute_reply.started":"2024-03-21T18:21:30.994187Z","shell.execute_reply":"2024-03-21T18:21:31.125636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:21:31.129081Z","iopub.execute_input":"2024-03-21T18:21:31.129382Z","iopub.status.idle":"2024-03-21T18:22:35.411881Z","shell.execute_reply.started":"2024-03-21T18:21:31.129357Z","shell.execute_reply":"2024-03-21T18:22:35.410827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install albumentations\n\nimport albumentations as albu\n\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, batch_size=32, shuffle=False, augment=False, mode='train', specs=spectrograms): \n\n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.mode = mode\n        self.specs = specs\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = int(np.ceil(len(self.data) / self.batch_size))\n        return ct\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(indexes)\n        if self.augment:\n            X = self.__augment_batch(X) \n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.data))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n                        \n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' \n        \n        #X los datos de entrada de los espectrogramas y Y contendrá las etiquetas\n        X = np.zeros((len(indexes), 128, 256, 4), dtype='float32')\n        y = np.zeros((len(indexes), 6), dtype='float32')\n        \n        # Se calcula la posición de inicio del espectrograma y se divide por 4,\n        # ya que  genera 4 imágenes de espectrograma como una imagen de 4 canales \n        # de tamaño 128x256x4 por muestra de tren. \n        for j, i in enumerate(indexes):\n            row = self.data.iloc[i]\n            if self.mode == 'test': \n                r = 0\n            else: \n                r = int((row['min'] + row['max']) // 4)\n\n                # Se itera sobre los cuatro canales de los espectrogramas \n                # para seleccionar, transformar y preprocesar cada uno de ellos.\n            for k in range(4):\n                # Se extraen 300 filas del espectrograma, y se seleccionan \n                # 100 columnas a partir de la posición r para obtener un\n                # espectrograma cuadrado de tamaño 300x100.\n                img = self.specs[row.spec_id][r:r+300, k*100:(k+1)*100].T\n                \n                # LOG TRANSFORM SPECTROGRAM\n                img = np.clip(img, np.exp(-4), np.exp(8))\n                img = np.log(img)\n                \n                # STANDARDIZE PER IMAGE\n                ep = 1e-6\n                m = np.nanmean(img.flatten())\n                s = np.nanstd(img.flatten())\n                img = (img - m) / (s + ep)\n                img = np.nan_to_num(img, nan=0.0)\n                \n                # CROP TO 256 TIME STEPS\n                X[j, 14:-14, :, k] = img[:, 22:-22] / 2.0\n        \n            if self.mode != 'test':\n                y[j] = row[TARGETS]\n            \n        return X, y\n    \n    def __random_transform(self, img):\n        composition = albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            # Add more transformations here if needed\n        ])\n        return composition(image=img)['image']\n            \n    def __augment_batch(self, img_batch):\n        for i in range(img_batch.shape[0]):\n            img_batch[i] = self.__random_transform(img_batch[i])\n        return img_batch\n","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:35.413552Z","iopub.execute_input":"2024-03-21T18:22:35.413999Z","iopub.status.idle":"2024-03-21T18:22:51.735625Z","shell.execute_reply.started":"2024-03-21T18:22:35.413961Z","shell.execute_reply":"2024-03-21T18:22:51.734499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gen = DataGenerator(train, batch_size=32, shuffle=False)\n\nROWS = 2\nCOLS = 2  \nBATCHES = 1\n\nfor i, (x, y) in enumerate(gen):\n    plt.figure(figsize=(20, 8))\n    for j in range(ROWS):\n        for k in range(COLS):\n            plt.subplot(ROWS, COLS, j * COLS + k + 1)\n            t = y[j * COLS + k]\n            img = x[j * COLS + k, :, :, 0][::-1, ]  # Seleccionamos solo el primer canal del espectrograma\n            mn = img.flatten().min()\n            mx = img.flatten().max()\n            img = (img - mn) / (mx - mn)  # Normalizamos la imagen\n            plt.imshow(img)\n            tars = f'[{t[0]:0.2f}'\n            for s in t[1:]:\n                tars += f', {s:0.2f}'\n            eeg = train.eeg_id.values[i * 32 + j * COLS + k]\n            plt.title(f'EEG = {eeg}\\nTarget = {tars}', size=12)\n            plt.yticks([])\n            plt.ylabel('Frequencies (Hz)', size=14)\n            plt.xlabel('Time (sec)', size=16)\n    plt.show()\n    if i == BATCHES - 1:\n        break\n","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:51.737199Z","iopub.execute_input":"2024-03-21T18:22:51.737823Z","iopub.status.idle":"2024-03-21T18:22:52.950219Z","shell.execute_reply.started":"2024-03-21T18:22:51.737790Z","shell.execute_reply":"2024-03-21T18:22:52.949194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for key, value in spectrograms.items():\n#    print(f\"Forma de {key}: {value.shape}\")","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-21T18:22:52.951514Z","iopub.execute_input":"2024-03-21T18:22:52.951821Z","iopub.status.idle":"2024-03-21T18:22:52.955932Z","shell.execute_reply.started":"2024-03-21T18:22:52.951795Z","shell.execute_reply":"2024-03-21T18:22:52.954886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nimport tensorflow as tf\n\ndef build_model():\n    inp = tf.keras.Input(shape=(128, 256, 4))  # Ajustamos la forma de entrada para coincidir con el modelo de CNN\n\n    # RESHAPE INPUT 128x256x4 => 512x512x3 MONOTONE IMAGE\n    x = tf.keras.layers.Concatenate(axis=3)([inp, inp, inp])  # Hacer una imagen monocromática 512x512x3\n\n    return tf.keras.Model(inputs=inp, outputs=x)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:52.957226Z","iopub.execute_input":"2024-03-21T18:22:52.957531Z","iopub.status.idle":"2024-03-21T18:22:52.974538Z","shell.execute_reply.started":"2024-03-21T18:22:52.957504Z","shell.execute_reply":"2024-03-21T18:22:52.973484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Input\n\ndef build_model():\n    inp = Input(shape=(128, 256, 4))  # Definir la entrada del modelo\n    \n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu', padding='same'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(64, activation='relu'),\n        Dense(6, activation='softmax')\n    ])\n    \n    opt = tf.keras.optimizers.Adam(learning_rate=1e-3)\n    loss = tf.keras.losses.CategoricalCrossentropy()\n\n    model.compile(loss=loss, optimizer=opt, metrics=['accuracy'])\n    \n    return model\n\n# Construir y compilar el modelo\nmodel = build_model()\n\n# Resumen del modelo\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:52.976097Z","iopub.execute_input":"2024-03-21T18:22:52.976916Z","iopub.status.idle":"2024-03-21T18:22:53.386967Z","shell.execute_reply.started":"2024-03-21T18:22:52.976878Z","shell.execute_reply":"2024-03-21T18:22:53.385979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Construir el modelo\n#model.build(input_shape=(None, 128, 256, 4))\n\n# Resumen del modelo\n#model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:53.389969Z","iopub.execute_input":"2024-03-21T18:22:53.390289Z","iopub.status.idle":"2024-03-21T18:22:53.394596Z","shell.execute_reply.started":"2024-03-21T18:22:53.390263Z","shell.execute_reply":"2024-03-21T18:22:53.393657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfrom scipy.interpolate import RectBivariateSpline\n\n# Calcular la forma máxima entre todos los espectrogramas\nshapes = [spectrogram.shape for spectrogram in spectrograms.values()]\nmax_shape = np.max(shapes, axis=0)\n\ndef resize_spectrogram(spectrogram, target_shape):\n    x = np.arange(spectrogram.shape[1])\n    y = np.arange(spectrogram.shape[0])\n    interpolator = RectBivariateSpline(y, x, spectrogram)\n    x_new = np.linspace(0, spectrogram.shape[1] - 1, target_shape[1])\n    y_new = np.linspace(0, spectrogram.shape[0] - 1, target_shape[0])\n    resized_spectrogram = interpolator(y_new, x_new)\n    return resized_spectrogram\n\nfor id, spectrogram in spectrograms.items():\n    # Redimensionar el espectrograma y obtener su forma\n    resized_spectrogram = resize_spectrogram(spectrogram, max_shape)\n    print(f\"Forma de {id} (espectrograma redimensionado): {resized_spectrogram.shape}\")\n    \n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:53.395871Z","iopub.execute_input":"2024-03-21T18:22:53.396243Z","iopub.status.idle":"2024-03-21T18:22:53.408280Z","shell.execute_reply.started":"2024-03-21T18:22:53.396210Z","shell.execute_reply":"2024-03-21T18:22:53.407244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Obtener los espectrogramas redimensionados\nX = np.array(list(spectrograms.values()))\nprint(\"Forma de X (espectrogramas redimensionados):\", X.shape)\n\n# División de los datos en entrenamiento y prueba\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Imprimir las formas de los conjuntos de entrenamiento y prueba\nprint(\"Forma de X_train (espectrogramas de entrenamiento):\", X_train.shape)\nprint(\"Forma de X_test (espectrogramas de prueba):\", X_test.shape)\n\n# Entrenamiento del modelo\nhistory = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_split=0.2)\n\n# Evaluación del modelo\ntest_loss, test_acc = model.evaluate(X_test, y_test)\nprint('Test accuracy:', test_acc)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:53.409597Z","iopub.execute_input":"2024-03-21T18:22:53.409924Z","iopub.status.idle":"2024-03-21T18:22:53.419880Z","shell.execute_reply.started":"2024-03-21T18:22:53.409898Z","shell.execute_reply":"2024-03-21T18:22:53.418869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definir el generador de datos de entrenamiento\ntrain_gen = DataGenerator(train, batch_size=32, shuffle=True, augment=True, mode='train')\n\n# Entrenar el modelo sin datos de validación\nhistory = model.fit(train_gen, epochs=8, batch_size=32)\n\n# Mostrar la evolución de la pérdida y la precisión durante el entrenamiento\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Metric')\nplt.legend()\nplt.show()\n\n# Guardar el modelo entrenado\nmodel.save('trained_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-21T18:22:53.421257Z","iopub.execute_input":"2024-03-21T18:22:53.421549Z","iopub.status.idle":"2024-03-21T18:33:06.262572Z","shell.execute_reply.started":"2024-03-21T18:22:53.421524Z","shell.execute_reply":"2024-03-21T18:33:06.261561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}