{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python -m pip install gwpy\n!pip install astropy","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-15T11:58:27.662305Z","iopub.execute_input":"2023-07-15T11:58:27.662700Z","iopub.status.idle":"2023-07-15T11:58:48.480923Z","shell.execute_reply.started":"2023-07-15T11:58:27.662667Z","shell.execute_reply":"2023-07-15T11:58:48.479706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import mixed_precision\nfrom gwpy.timeseries import TimeSeries\nfrom gwpy.plot import Plot\nimport numpy as np\nfrom scipy import signal\nfrom sklearn.preprocessing import MinMaxScaler\nfrom PIL import Image\nfrom matplotlib import pyplot as plt\nfrom matplotlib.gridspec import GridSpec\nimport pandas as pd","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-15T11:58:48.483135Z","iopub.execute_input":"2023-07-15T11:58:48.483466Z","iopub.status.idle":"2023-07-15T11:59:31.276746Z","shell.execute_reply.started":"2023-07-15T11:58:48.483434Z","shell.execute_reply":"2023-07-15T11:59:31.275646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to get hardware strategy\ndef get_hardware_strategy():\n    try:\n        # TPU detection. No parameters necessary if TPU_NAME environment variable is\n        # set: this is always the case on Kaggle.\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print('Running on TPU ', tpu.master())\n    except ValueError:\n        tpu = None\n\n    if tpu:\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        policy = mixed_precision.Policy('mixed_bfloat16')\n        mixed_precision.set_global_policy(policy)\n        tf.config.optimizer.set_jit(True)\n    else:\n        # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n        strategy = tf.distribute.get_strategy()\n\n    print(\"REPLICAS: \", strategy.num_replicas_in_sync)\n    return tpu, strategy\n\ntpu, strategy = get_hardware_strategy()","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:00:09.063753Z","iopub.execute_input":"2023-07-15T12:00:09.064122Z","iopub.status.idle":"2023-07-15T12:00:17.662662Z","shell.execute_reply.started":"2023-07-15T12:00:09.064090Z","shell.execute_reply":"2023-07-15T12:00:17.660951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/g2net-gravitational-wave-detection/training_labels.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/g2net-gravitational-wave-detection/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:00:00.358810Z","iopub.execute_input":"2023-07-15T12:00:00.359765Z","iopub.status.idle":"2023-07-15T12:00:01.186556Z","shell.execute_reply.started":"2023-07-15T12:00:00.359704Z","shell.execute_reply":"2023-07-15T12:00:01.185684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Q_RANGE = (16,32)\nF_RANGE = (30,400)\n\ndef id2path(idx,is_train=True):\n    path = '../input/g2net-gravitational-wave-detection'\n    if is_train:\n        path += '/train/'+idx[0]+'/'+idx[1]+'/'+idx[2]+'/'+idx+'.npy'\n    else:\n        path += '/test/'+idx[0]+'/'+idx[1]+'/'+idx[2]+'/'+idx+'.npy'\n    return path\n\ndef read_file(fname, is_train):\n    data = np.load(id2path(fname, is_train))\n    d1 = TimeSeries(data[0,:], sample_rate=2048)\n    d2 = TimeSeries(data[1,:], sample_rate=2048)\n    d3 = TimeSeries(data[2,:], sample_rate=2048)\n    return d1, d2, d3\n\n\ndef preprocess(d1, d2, d3, bandpass=False, lf=35, hf=350):\n    white_d1 = d1.whiten(window=(\"tukey\",0.2))\n    white_d2 = d2.whiten(window=(\"tukey\",0.2))\n    white_d3 = d3.whiten(window=(\"tukey\",0.2))\n    if bandpass: # bandpass filter\n        bp_d1 = white_d1.bandpass(lf, hf) \n        bp_d2 = white_d2.bandpass(lf, hf)\n        bp_d3 = white_d3.bandpass(lf, hf)\n        return bp_d1, bp_d2, bp_d3\n    else: # only whiten\n        return white_d1, white_d2, white_d3\n\ndef create_rgb(fname, is_train):\n    r1, r2, r3 = read_file(fname, is_train)\n    p1, p2, p3 = preprocess(r1, r2, r3)\n    hq1 = p1.q_transform(qrange=Q_RANGE, frange=F_RANGE, logf=True, whiten=False)\n    hq2 = p2.q_transform(qrange=Q_RANGE, frange=F_RANGE, logf=True, whiten=False)\n    hq3 = p3.q_transform(qrange=Q_RANGE, frange=F_RANGE, logf=True, whiten=False)\n    img = np.zeros([hq1.shape[0], hq1.shape[1], 3], dtype=np.uint8)\n    scaler = MinMaxScaler()\n    img[:,:,0] = 255*scaler.fit_transform(hq1)\n    img[:,:,1] = 255*scaler.fit_transform(hq2)\n    img[:,:,2] = 255*scaler.fit_transform(hq3)\n    img_pil = Image.fromarray(img).rotate(90, expand=1).resize((60,60))\n    return np.array(img_pil)\n\n# Ejemplo de uso:\nimage = create_rgb(sample_submission[\"id\"][109], is_train = False)\nprint(image.shape)\nplt.imshow(image)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:00:35.594575Z","iopub.execute_input":"2023-07-15T12:00:35.595306Z","iopub.status.idle":"2023-07-15T12:00:36.111034Z","shell.execute_reply.started":"2023-07-15T12:00:35.595273Z","shell.execute_reply":"2023-07-15T12:00:36.109867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_sample_target = train_labels[train_labels[\"target\"] == 1].reset_index(drop = True)\ntargets = data_sample_target[\"id\"].iloc[48:50]\nno_targets = train_labels[train_labels[\"target\"] == 1][\"id\"].head(2)\nfig = plt.figure(figsize=(15, 10))\ngs = GridSpec(4, 4, figure=fig)\n\nfor i, (target, no_target) in enumerate(zip(targets, no_targets)):\n    # Subplot para GW Found (gráfico de líneas)\n    ax1 = fig.add_subplot(gs[i*2, 0])\n    ax1.plot(np.load(id2path(target))[1, :], color='blue')\n    ax1.set_title(\"GW Found - Ligo Hanford\")\n    ax1.set_axis_off()\n\n    ax2 = fig.add_subplot(gs[i*2, 1])\n    ax2.plot(np.load(id2path(target))[2, :], color='blue')\n    ax2.set_title(\"GW Found - Virgo\")\n    ax2.set_axis_off()\n\n    ax3 = fig.add_subplot(gs[i*2, 2])\n    ax3.plot(np.load(id2path(target))[0, :], color='blue')\n    ax3.set_title(\"GW Found - Ligo Livingstone\")\n    ax3.set_axis_off()\n    \n    ax4 = fig.add_subplot(gs[i*2, 3])\n    ax4.imshow(create_rgb(target, is_train = True))\n    ax4.set_title(\"GW Found - Spectrogram\")\n    ax4.set_axis_off()\n    \n\n    # Subplot para GW Not Found (gráfico de líneas)\n    ax5 = fig.add_subplot(gs[i*2 + 1, 0])\n    ax5.plot(np.load(id2path(no_target))[1, :], color='red')\n    ax5.set_title(\"GW Not Found - Ligo Hanford\")\n    ax5.set_axis_off()\n\n    ax6 = fig.add_subplot(gs[i*2 + 1, 1])\n    ax6.plot(np.load(id2path(no_target))[2, :], color='red')\n    ax6.set_title(\"GW Not Found - Virgo\")\n    ax6.set_axis_off()\n\n    ax7 = fig.add_subplot(gs[i*2 + 1, 2])\n    ax7.plot(np.load(id2path(no_target))[0, :], color='red')\n    ax7.set_title(\"GW Not Found - Ligo Livingstone\")\n    ax7.set_axis_off()\n\n    ax8 = fig.add_subplot(gs[i*2 + 1, 3])\n    ax8.imshow(create_rgb(no_target, is_train = True))\n    ax8.set_title(\"GW Not Found - Spectrogram\")\n    ax8.set_axis_off()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:01:03.380502Z","iopub.execute_input":"2023-07-15T12:01:03.380931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nclass G2NetDataset(tf.keras.utils.Sequence):\n    def __init__(self, data, y = None, batch_size = 256, shuffle = True):\n        \n        self.data = data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        \n        if y is not None:\n            self.is_train = True\n        else:\n            self.is_train = False\n        self.y = y\n        \n    def __len__(self):\n        return math.ceil(len(self.data) / self.batch_size)\n    \n    def __getitem__(self, ids):\n        batch_data = self.data[ids * self.batch_size : (ids + 1) * self.batch_size]\n        \n        if y is not None:\n            batch_y = self.y[ids * self.batch_size : (ids + 1) * self.batch_size]\n            \n        batch_x = np.array([create_rgb(x, self.is_train) for x in batch_data])\n        batch_x = np.stack(batch_x)\n        \n        if self.is_train:\n            return batch_x, batch_y\n        else:\n            return batch_x\n        \n        \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.data, self.y))\n            shuffle(ids_y)\n            shuffle(ids_y)\n            self.data, self.y = list(zip(*ids_y))  \n        \n            \n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:01:09.262913Z","iopub.execute_input":"2023-07-15T12:01:09.263288Z","iopub.status.idle":"2023-07-15T12:01:09.276936Z","shell.execute_reply.started":"2023-07-15T12:01:09.263257Z","shell.execute_reply":"2023-07-15T12:01:09.275850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split as tts","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:01:10.560429Z","iopub.execute_input":"2023-07-15T12:01:10.560786Z","iopub.status.idle":"2023-07-15T12:01:10.681113Z","shell.execute_reply.started":"2023-07-15T12:01:10.560757Z","shell.execute_reply":"2023-07-15T12:01:10.680154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idx =  train_labels['id'].values\ny = train_labels['target'].values\ntest_idx = sample_submission['id'].values\n\nx_train,x_valid,y_train,y_valid = tts(train_idx,y,test_size=0.05,random_state=42,stratify=y)\n\ntrain_dataset = G2NetDataset(x_train,y_train)\nvalid_dataset = G2NetDataset(x_valid,y_valid)\ntest_dataset = G2NetDataset(test_idx)","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:01:11.347496Z","iopub.execute_input":"2023-07-15T12:01:11.347814Z","iopub.status.idle":"2023-07-15T12:01:11.619691Z","shell.execute_reply.started":"2023-07-15T12:01:11.347786Z","shell.execute_reply":"2023-07-15T12:01:11.618690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U efficientnet","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:01:13.930188Z","iopub.execute_input":"2023-07-15T12:01:13.930547Z","iopub.status.idle":"2023-07-15T12:01:22.480238Z","shell.execute_reply.started":"2023-07-15T12:01:13.930510Z","shell.execute_reply":"2023-07-15T12:01:22.478954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport efficientnet.keras as efn\n\ndef get_model():\n    inp = tf.keras.layers.Input(shape=(60, 60, 3))\n    # Usamos el argumento pooling para aplicar el GlobalAveragePooling2D dentro del modelo EfficientNetB7\n    x = efn.EfficientNetB7(include_top=False, weights='imagenet', pooling='avg')(inp)\n    # Añadimos una capa de dropout para regularizar el modelo y evitar el sobreajuste\n    x = tf.keras.layers.Dropout(0.2)(x)\n    output = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    model = tf.keras.models.Model(inputs=inp, outputs=output)\n    \n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss=tf.keras.losses.BinaryCrossentropy(),\n        metrics=[tf.keras.metrics.AUC()],\n        \n    )\n    \n    return model\n\n# Crear el modelo dentro de un alcance estratégico\nwith strategy.scope():\n    model = get_model()\n\n    # Entrenar el modelo con ReduceLROnPlateau como callback\n    model.fit(\n        train_dataset,\n        epochs=1,\n        validation_data=valid_dataset,\n        batch_size=64  # Aumenta el tamaño del lote aquí\n    )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-15T12:19:25.097159Z","iopub.execute_input":"2023-07-15T12:19:25.097884Z","iopub.status.idle":"2023-07-15T12:36:13.747021Z","shell.execute_reply.started":"2023-07-15T12:19:25.097839Z","shell.execute_reply":"2023-07-15T12:36:13.744103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history[\"history\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_dataset)\npreds = preds.reshape(-1)\n\nsubmission = pd.DataFrame({'id':sample_submission['id'],'target':preds})\nsubmission.to_csv('submission.csv',index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}