{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Dataset - Specific Installations\n# Reference link: https://gwpy.github.io/docs/latest/plot/index.html#multi-data-plots%20-\n\n!python -m pip install gwpy\n!pip install astropy\n!pip install nnAudio","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:52:42.206102Z","iopub.execute_input":"2022-12-21T05:52:42.207059Z","iopub.status.idle":"2022-12-21T05:53:18.604763Z","shell.execute_reply.started":"2022-12-21T05:52:42.206969Z","shell.execute_reply":"2022-12-21T05:53:18.603386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Basic libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\nfrom scipy import signal\n\nfrom glob import glob\nfrom PIL import Image\n\n#Dataset specific libraries\nfrom gwpy.timeseries import TimeSeries\nfrom gwpy.plot import Plot\n\n#Libraries for model building and training\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Conv1D, MaxPool1D, BatchNormalization\nfrom tensorflow.keras.optimizers import RMSprop, Adam\n\n#Other Miscellaneous libraries\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport torch\nimport random\nfrom nnAudio.Spectrogram import CQT1992v2\n\nplt.style.use('ggplot')\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:53:18.607166Z","iopub.execute_input":"2022-12-21T05:53:18.608687Z","iopub.status.idle":"2022-12-21T05:53:28.727553Z","shell.execute_reply.started":"2022-12-21T05:53:18.608639Z","shell.execute_reply":"2022-12-21T05:53:28.726586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the data contents\n\nfile = \"../input/g2net-gravitational-wave-detection/train/0/0/0/00000e74ad.npy\"\ndata = np.load(file)\nprint(data.shape)\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:53:28.729049Z","iopub.execute_input":"2022-12-21T05:53:28.729823Z","iopub.status.idle":"2022-12-21T05:53:28.753695Z","shell.execute_reply.started":"2022-12-21T05:53:28.729774Z","shell.execute_reply":"2022-12-21T05:53:28.752613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_timeseries_from_file(file_name):\n  data = np.load(file_name)\n  ts1 = TimeSeries(data[0,:], sample_rate=2048)\n  ts2 = TimeSeries(data[1,:], sample_rate=2048)\n  ts3 = TimeSeries(data[2,:], sample_rate=2048)\n  return ts1, ts2, ts3\n\ndef plot_timeseries(t1, t2, t3):\n  plot = Plot(t1, t2, t3, separate=True, sharex=True, figsize=[20, 12])\n  ax = plot.gca()\n  ax.set_xlim(0, 2)\n  ax.set_xlabel('Time [s]')\n  plt.show()\n  \nfile = \"../input/g2net-gravitational-wave-detection/train/0/0/0/00000e74ad.npy\"\n  \nt1, t2, t3 = get_timeseries_from_file(file)\n\n# Plotting the TimeSeries\nplot_timeseries(t1, t2, t3)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:53:28.756432Z","iopub.execute_input":"2022-12-21T05:53:28.757129Z","iopub.status.idle":"2022-12-21T05:53:33.341302Z","shell.execute_reply.started":"2022-12-21T05:53:28.757093Z","shell.execute_reply":"2022-12-21T05:53:33.340348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"../input/g2net-gravitational-wave-detection/training_labels.csv\")\nprint(train_labels.head())","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:53:33.342452Z","iopub.execute_input":"2022-12-21T05:53:33.342790Z","iopub.status.idle":"2022-12-21T05:53:33.721371Z","shell.execute_reply.started":"2022-12-21T05:53:33.342754Z","shell.execute_reply":"2022-12-21T05:53:33.720332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths = glob(\"../input/g2net-gravitational-wave-detection/train/*/*/*/*\")\n\nids_from_npy_files = [path.split(\"/\")[-1].split(\".\")[0] for path in train_paths]\n\ndf_path_id = pd.DataFrame({'path': train_paths, 'id':ids_from_npy_files})\ndf_path_id.head()\n\ndf_train = pd.merge(left=train_labels, right=df_path_id, on='id')\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:53:33.722908Z","iopub.execute_input":"2022-12-21T05:53:33.723252Z","iopub.status.idle":"2022-12-21T05:56:13.406395Z","shell.execute_reply.started":"2022-12-21T05:53:33.723217Z","shell.execute_reply":"2022-12-21T05:56:13.405504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0 = df_train[df_train.target == 0]\ndf_train_1 = df_train[df_train.target == 1]\n\nsns.countplot(x = 'target' , data=train_labels)\nplt.title('0 vs 1')","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:13.407767Z","iopub.execute_input":"2022-12-21T05:56:13.408129Z","iopub.status.idle":"2022-12-21T05:56:13.810318Z","shell.execute_reply.started":"2022-12-21T05:56:13.408092Z","shell.execute_reply":"2022-12-21T05:56:13.809361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gwaves = np.load(df_train.loc[2, 'path'])\nprint(gwaves.shape)\ngwaves_stacked = np.hstack(gwaves)\nprint(gwaves_stacked.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:13.811626Z","iopub.execute_input":"2022-12-21T05:56:13.812636Z","iopub.status.idle":"2022-12-21T05:56:13.843009Z","shell.execute_reply.started":"2022-12-21T05:56:13.812596Z","shell.execute_reply":"2022-12-21T05:56:13.841963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def spectrogram(\n    gwaves,\n    transform=CQT1992v2(sr=2048, fmin=20, fmax=1024, hop_length=64),\n):\n    stacked_waves_from_each_file = np.hstack(gwaves)\n    stacked_waves_from_each_file = stacked_waves_from_each_file / np.max(\n        stacked_waves_from_each_file\n    )\n    stacked_waves_from_each_file = torch.from_numpy(stacked_waves_from_each_file).float()\n    cqt_image = transform(stacked_waves_from_each_file)\n    return cqt_image\n\n\nfor i in range(5):\n  cqt_img = spectrogram(np.load(df_train.loc[i, 'path']))\n  target = df_train.loc[i, 'target']\n  plt.imshow(cqt_img[0])\n  plt.title(f\"target: {target}\")\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:13.844598Z","iopub.execute_input":"2022-12-21T05:56:13.844959Z","iopub.status.idle":"2022-12-21T05:56:15.696282Z","shell.execute_reply.started":"2022-12-21T05:56:13.844922Z","shell.execute_reply":"2022-12-21T05:56:15.695320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_PATH = '../input/g2net-q-transform-69x65/images'\n\nclass DataGenerator(Sequence):\n    \n    def __init__(self, df, batch_size=32,n_batches = None, img_size=(69,65), shuffle=True, is_train=True):\n        self.df = df\n        self.n = len(df)\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.shuffle = shuffle\n        self.is_train = is_train\n        self.on_epoch_end()\n        self.n_batches = n_batches\n        \n    def on_epoch_end(self):\n        self.indices = np.arange(self.n)\n        if self.shuffle == True:\n            np.random.shuffle(self.indices)   \n    \n    def __len__(self):\n        if self.n_batches is None:\n            return math.ceil( self.n / self.batch_size )\n        return self.n_batches\n    \n    def __getitem__(self, batch_index):\n        start = batch_index * self.batch_size\n        end = (batch_index + 1) * self.batch_size\n        indices = self.indices[start:end]\n        \n        return self.__data_generation(indices)\n    \n    def __data_generation(self, batch_indices):\n        batch_size = len(batch_indices)\n        \n        X = np.zeros(shape=(batch_size, self.img_size[0], self.img_size[1], 3))\n        y = np.zeros(batch_size)\n        \n        for i, idx in enumerate(batch_indices):\n            ID = self.df.id.values[idx]\n            y[i] = self.df.target.values[idx]\n            \n            SET = 'train' if self.is_train else 'test'\n            path = f'{SPEC_PATH}/{SET}/{ID}.npy'\n            data_array = np.load(path)\n            \n            X[i,:,:,:] = data_array\n            \n        return X, y\n    \n\nGENERATOR_TEST = True\n\nif GENERATOR_TEST:\n    temp_gen = DataGenerator(train_labels, batch_size=8, shuffle=False)\n    X,y = temp_gen.__getitem__(0)\n\n    print(X.shape)\n    print(y)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:15.699942Z","iopub.execute_input":"2022-12-21T05:56:15.700610Z","iopub.status.idle":"2022-12-21T05:56:15.767245Z","shell.execute_reply.started":"2022-12-21T05:56:15.700571Z","shell.execute_reply":"2022-12-21T05:56:15.766159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[16,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:15.770161Z","iopub.execute_input":"2022-12-21T05:56:15.770457Z","iopub.status.idle":"2022-12-21T05:56:15.778392Z","shell.execute_reply.started":"2022-12-21T05:56:15.770426Z","shell.execute_reply":"2022-12-21T05:56:15.777417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U efficientnet -qq","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:15.780501Z","iopub.execute_input":"2022-12-21T05:56:15.781485Z","iopub.status.idle":"2022-12-21T05:56:26.205838Z","shell.execute_reply.started":"2022-12-21T05:56:15.781450Z","shell.execute_reply":"2022-12-21T05:56:26.204494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn\nimport tensorflow as tf\nfrom tensorflow.keras.layers import *\n\nmodel = efn.EfficientNetB7(input_shape=(69,65,3), include_top=False, weights='imagenet')\nmodel.trainable = False\nopt = tf.keras.optimizers.Adam(.0001)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:56:26.208082Z","iopub.execute_input":"2022-12-21T05:56:26.208496Z","iopub.status.idle":"2022-12-21T05:57:06.997360Z","shell.execute_reply.started":"2022-12-21T05:56:26.208452Z","shell.execute_reply":"2022-12-21T05:57:06.996356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train_labels, test_size=0.15)\ntrain_loader = DataGenerator(train_df, batch_size=64, shuffle=True)\nvalid_loader = DataGenerator(valid_df, batch_size=64, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:57:06.998989Z","iopub.execute_input":"2022-12-21T05:57:06.999352Z","iopub.status.idle":"2022-12-21T05:57:07.092234Z","shell.execute_reply.started":"2022-12-21T05:57:06.999313Z","shell.execute_reply":"2022-12-21T05:57:07.091270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = efn.EfficientNetB7(input_shape=(69,65,3), include_top=False, weights='imagenet')\nmodel.trainable = False\ncnn = Sequential([model,GlobalAveragePooling2D(),Dropout(0.2),Dense(64, activation='relu'),BatchNormalization(),Dense(1, activation='sigmoid')])\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:57:07.093939Z","iopub.execute_input":"2022-12-21T05:57:07.094309Z","iopub.status.idle":"2022-12-21T05:57:15.013711Z","shell.execute_reply.started":"2022-12-21T05:57:07.094269Z","shell.execute_reply":"2022-12-21T05:57:15.011746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nopt = tf.keras.optimizers.Adam(0.01)\ncnn.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])\nh1 = cnn.fit(train_loader, epochs=5, validation_data=valid_loader, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T05:57:15.015144Z","iopub.execute_input":"2022-12-21T05:57:15.015538Z","iopub.status.idle":"2022-12-21T10:49:25.295974Z","shell.execute_reply.started":"2022-12-21T05:57:15.015500Z","shell.execute_reply":"2022-12-21T10:49:25.292082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:49:25.303823Z","iopub.execute_input":"2022-12-21T10:49:25.304116Z","iopub.status.idle":"2022-12-21T10:49:26.345419Z","shell.execute_reply.started":"2022-12-21T10:49:25.304088Z","shell.execute_reply":"2022-12-21T10:49:26.344344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.trainable = True\nopt = tf.keras.optimizers.Adam(.0001)\ncnn.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])\nh2 = cnn.fit(train_loader, epochs=5, validation_data=valid_loader, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:49:26.346697Z","iopub.execute_input":"2022-12-21T10:49:26.347057Z","iopub.status.idle":"2022-12-21T15:15:35.489574Z","shell.execute_reply.started":"2022-12-21T10:49:26.347021Z","shell.execute_reply":"2022-12-21T15:15:35.487052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h2.history['auc'] = h2.history['auc_1']\nh2.history['val_auc'] = h2.history['val_auc_1']","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:15:35.497490Z","iopub.execute_input":"2022-12-21T15:15:35.499337Z","iopub.status.idle":"2022-12-21T15:15:35.509592Z","shell.execute_reply.started":"2022-12-21T15:15:35.499297Z","shell.execute_reply":"2022-12-21T15:15:35.508561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1,h2])\nvis_training(history, start = 5)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:15:35.511075Z","iopub.execute_input":"2022-12-21T15:15:35.511516Z","iopub.status.idle":"2022-12-21T15:15:36.462041Z","shell.execute_reply.started":"2022-12-21T15:15:35.511482Z","shell.execute_reply":"2022-12-21T15:15:36.461050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/g2net-q-transform-69x65/images/sample_submission.csv')\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:15:36.463778Z","iopub.execute_input":"2022-12-21T15:15:36.464440Z","iopub.status.idle":"2022-12-21T15:15:36.703626Z","shell.execute_reply.started":"2022-12-21T15:15:36.464385Z","shell.execute_reply":"2022-12-21T15:15:36.702472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataGenerator(test, batch_size=64, shuffle=False, is_train=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:15:36.705105Z","iopub.execute_input":"2022-12-21T15:15:36.706132Z","iopub.status.idle":"2022-12-21T15:15:36.711560Z","shell.execute_reply.started":"2022-12-21T15:15:36.706095Z","shell.execute_reply":"2022-12-21T15:15:36.710309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prob = cnn.predict(test_loader)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:15:36.713050Z","iopub.execute_input":"2022-12-21T15:15:36.714103Z","iopub.status.idle":"2022-12-21T15:54:04.822202Z","shell.execute_reply.started":"2022-12-21T15:15:36.714066Z","shell.execute_reply":"2022-12-21T15:54:04.817699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('Result.h')","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:54:04.831604Z","iopub.execute_input":"2022-12-21T15:54:04.831990Z","iopub.status.idle":"2022-12-21T15:56:20.622992Z","shell.execute_reply.started":"2022-12-21T15:54:04.831953Z","shell.execute_reply":"2022-12-21T15:56:20.620977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/g2net-q-transform-69x65/images/sample_submission.csv')\nsubmission.target = test_prob[:,0]\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-12-21T15:56:20.644973Z","iopub.execute_input":"2022-12-21T15:56:20.646443Z","iopub.status.idle":"2022-12-21T15:56:20.931453Z","shell.execute_reply.started":"2022-12-21T15:56:20.646385Z","shell.execute_reply":"2022-12-21T15:56:20.930365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('Result.csv', header=True, index= False)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T16:14:09.529565Z","iopub.execute_input":"2022-12-21T16:14:09.529946Z","iopub.status.idle":"2022-12-21T16:14:09.890220Z","shell.execute_reply.started":"2022-12-21T16:14:09.529914Z","shell.execute_reply":"2022-12-21T16:14:09.889169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink('Result.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-21T16:14:52.232479Z","iopub.execute_input":"2022-12-21T16:14:52.233467Z","iopub.status.idle":"2022-12-21T16:14:52.241897Z","shell.execute_reply.started":"2022-12-21T16:14:52.233397Z","shell.execute_reply":"2022-12-21T16:14:52.240738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}