{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **G2Net Gravitational Wave Detection**","metadata":{}},{"cell_type":"markdown","source":"### **Setup**\nImporting necessary modules","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom glob import glob\nimport random\nfrom colorama import Fore, Back, Style\n\nplt.style.use('ggplot')\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T04:15:31.617456Z","iopub.execute_input":"2022-07-26T04:15:31.618719Z","iopub.status.idle":"2022-07-26T04:15:32.839860Z","shell.execute_reply.started":"2022-07-26T04:15:31.618625Z","shell.execute_reply":"2022-07-26T04:15:32.838778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading the data","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv(\"../input/g2net-gravitational-wave-detection/training_labels.csv\")\n\nprint(Fore.BLUE + \"Dataset has \",Style.RESET_ALL + \"{} Observations\".format(labels.shape[0]))\n\nprint(Fore.GREEN + \"First 5 Observations:\",Style.RESET_ALL)\ndisplay(labels.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:15:32.845276Z","iopub.execute_input":"2022-07-26T04:15:32.845600Z","iopub.status.idle":"2022-07-26T04:15:33.304617Z","shell.execute_reply.started":"2022-07-26T04:15:32.845567Z","shell.execute_reply":"2022-07-26T04:15:33.303198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a training dataframe for all the available .npy files along with their path\n\n# path of the files\npaths = glob(\"../input/g2net-gravitational-wave-detection/train/*/*/*/*\")\n\n# list of ids of .npy files \nids = [path.split(\"/\")[-1].split(\".\")[0] for path in paths]\n\n# data frame containing paths and ids of .npy files \npath_df = pd.DataFrame({\"path\":paths,\"id\":ids})\n\n# merge the dataframe built above with the dataset having target\ntrain_df = pd.merge(left=labels,right=path_df,on=\"id\")\n\n# this would a comprehensive df which would include \"id\",\"target\" and \"path\" for each of the .npy file in train folder\ndisplay(train_df.head())\n\n# lets confirm whether the dataframe built above has expected no. of rows(560000)\nprint(Fore.BLUE + \"No.of rows in the merged dataframe:\",train_df.shape[0],Style.RESET_ALL)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:15:33.306097Z","iopub.execute_input":"2022-07-26T04:15:33.306950Z","iopub.status.idle":"2022-07-26T04:18:41.090181Z","shell.execute_reply.started":"2022-07-26T04:15:33.306915Z","shell.execute_reply":"2022-07-26T04:18:41.088989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# segregate dataframes for individual classes\ntarget_1 = train_df[train_df.target==1]\ntarget_0 = train_df[train_df.target==0]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.094311Z","iopub.execute_input":"2022-07-26T04:18:41.095145Z","iopub.status.idle":"2022-07-26T04:18:41.193838Z","shell.execute_reply.started":"2022-07-26T04:18:41.095099Z","shell.execute_reply":"2022-07-26T04:18:41.192343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Class Distribution**\n\nBoth the labels (target=0, target=1) have equal distribution in dataset","metadata":{}},{"cell_type":"code","source":"print(\"Class Distribution:\\n\",labels.target.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.195383Z","iopub.execute_input":"2022-07-26T04:18:41.196084Z","iopub.status.idle":"2022-07-26T04:18:41.209592Z","shell.execute_reply.started":"2022-07-26T04:18:41.196035Z","shell.execute_reply":"2022-07-26T04:18:41.208435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize class distribution\nsns.countplot(x=\"target\", data=labels)\nplt.title(\"Class Distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.211324Z","iopub.execute_input":"2022-07-26T04:18:41.212410Z","iopub.status.idle":"2022-07-26T04:18:41.455798Z","shell.execute_reply.started":"2022-07-26T04:18:41.212354Z","shell.execute_reply":"2022-07-26T04:18:41.454622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Data Analysis**\n\nSITE-1, SITE-2 & SITE-3 are 3 different serieses","metadata":{}},{"cell_type":"code","source":"# visualize the randomly selected series\ndef plot_series(series,plot,target):\n    if plot == \"box\" or plot == \"kde\":\n        plt.figure(figsize = (20,2))    \n    else:\n        plt.figure(figsize = (15,12))    \n    \n    for idx in range(3):\n        if plot == \"box\":\n            plt.subplot(1,3,idx+1)            \n            sns.boxplot(series[idx:idx+1],color = 'b')  \n            \n        elif plot == \"kde\":\n            plt.subplot(1,3,idx+1)            \n            sns.kdeplot(series[idx],color = 'r', shade=True,lw=2, alpha=0.5)\n        else:\n            plt.subplot(3,1,idx+1)            \n            plt.plot(series[idx:idx+1].T,color = 'g')\n            plt.title(\"\\nSite-\" + str(idx+1))        \n            \n    if plot == \"box\":    \n        plt.suptitle(\"Box Plots(target = \" + target + \")\")\n    elif plot == \"kde\":    \n        plt.suptitle(\"Probablity Distribution Plots(target = \" + target + \")\")\n    else:    \n        plt.suptitle(\"Time Distribution of Signals - Spans 2 sec, Sampled at 2,048 Hz(target = \" + target + \")\")\n\n        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.458351Z","iopub.execute_input":"2022-07-26T04:18:41.458856Z","iopub.status.idle":"2022-07-26T04:18:41.469191Z","shell.execute_reply.started":"2022-07-26T04:18:41.458807Z","shell.execute_reply":"2022-07-26T04:18:41.467811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pick a random series(target=1)\ntarget_1 = target_1.sample(1).path.values[0]\n\npos = np.load(target_1)\n\nprint(Fore.BLUE + \"Shape of the selected signal:\",pos.shape,Style.RESET_ALL)\nprint(\"\\n\\n\")\n\nplot_series(pos,\"time\",\"1\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.470948Z","iopub.execute_input":"2022-07-26T04:18:41.471457Z","iopub.status.idle":"2022-07-26T04:18:41.940358Z","shell.execute_reply.started":"2022-07-26T04:18:41.471424Z","shell.execute_reply":"2022-07-26T04:18:41.938991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pick a random series(target=1)\ntarget_0 = target_0.sample(1).path.values[0]\n\nneg = np.load(target_0)\nprint(Fore.BLUE + \"Shape of the selected signal:\",neg.shape,Style.RESET_ALL)\nprint(\"\\n\\n\")\n\nplot_series(neg,\"plot\",\"0\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:41.942214Z","iopub.execute_input":"2022-07-26T04:18:41.942581Z","iopub.status.idle":"2022-07-26T04:18:42.444520Z","shell.execute_reply.started":"2022-07-26T04:18:41.942541Z","shell.execute_reply":"2022-07-26T04:18:42.443389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have a huge time series of approximately 560,000 files each of which has dimensions of 3 * 4096.\n\nThe series with no signals have bigger fluctuations, whereas the series with signal absent have smaller and more consistent fluctuations.","metadata":{}},{"cell_type":"code","source":"# Probability Distribution plots for target == 1 (Signal is missing)\nplot_series(pos,\"box\",\"1\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:42.446266Z","iopub.execute_input":"2022-07-26T04:18:42.446968Z","iopub.status.idle":"2022-07-26T04:18:42.751932Z","shell.execute_reply.started":"2022-07-26T04:18:42.446924Z","shell.execute_reply":"2022-07-26T04:18:42.750592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Probability Distribution plots for target == 0 (Signal is missing)\nplot_series(neg,\"box\",\"0\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:42.753488Z","iopub.execute_input":"2022-07-26T04:18:42.753936Z","iopub.status.idle":"2022-07-26T04:18:43.073814Z","shell.execute_reply.started":"2022-07-26T04:18:42.753891Z","shell.execute_reply":"2022-07-26T04:18:43.072983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These 3 sites ahve fairly similar distribution for both class types.\n\nThird site seems to have difference in outliers which is the only visible difference.","metadata":{}},{"cell_type":"code","source":"# Probability Distribution plots for target == 1 (Signal is present)\nplot_series(pos,\"kde\",\"1\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:43.074938Z","iopub.execute_input":"2022-07-26T04:18:43.075791Z","iopub.status.idle":"2022-07-26T04:18:43.488132Z","shell.execute_reply.started":"2022-07-26T04:18:43.075728Z","shell.execute_reply":"2022-07-26T04:18:43.486587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Probability Distribution plots for target == 0 (Signal is missing)\nplot_series(neg,\"kde\",\"0\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:43.492095Z","iopub.execute_input":"2022-07-26T04:18:43.492491Z","iopub.status.idle":"2022-07-26T04:18:44.070814Z","shell.execute_reply.started":"2022-07-26T04:18:43.492456Z","shell.execute_reply":"2022-07-26T04:18:44.069684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"KDE plots for both classes looks almost similar for SITE-3; SITE-2  has little bit more variation for target=0 whereas SITE-1 has more variation for target=1.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.utils import Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv1D, MaxPool1D, BatchNormalization\nfrom tensorflow.keras.optimizers import RMSprop,Adam\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:44.072075Z","iopub.execute_input":"2022-07-26T04:18:44.073011Z","iopub.status.idle":"2022-07-26T04:18:51.362578Z","shell.execute_reply.started":"2022-07-26T04:18:44.072978Z","shell.execute_reply":"2022-07-26T04:18:51.361027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Data Generator**\n\nTo feed real-time data to Keras Model.","metadata":{}},{"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, data, batch_size):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.data = data\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, 3, 4096))\n        y = np.zeros((self.batch_size, 1))\n        for i, ID in enumerate(list_IDs_temp):\n            id_ = self.data.loc[ID, 'id']\n            file = id_+'.npy'\n            path_in = '/'.join([self.path, id_[0], id_[1], id_[2]])+'/'\n            data_array = np.load(path_in+file)\n            data_array = (data_array-data_array.mean())/data_array.std()\n            X[i, ] = data_array\n            y[i, ] = self.data.loc[ID, 'target']\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:51.364060Z","iopub.execute_input":"2022-07-26T04:18:51.364763Z","iopub.status.idle":"2022-07-26T04:18:51.376118Z","shell.execute_reply.started":"2022-07-26T04:18:51.364699Z","shell.execute_reply":"2022-07-26T04:18:51.374981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/g2net-gravitational-wave-detection/sample_submission.csv')\ntrain_idx =  labels['id'].values\ny = labels['target'].values\ntest_idx = sample_submission['id'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:51.377872Z","iopub.execute_input":"2022-07-26T04:18:51.378389Z","iopub.status.idle":"2022-07-26T04:18:51.605879Z","shell.execute_reply.started":"2022-07-26T04:18:51.378357Z","shell.execute_reply":"2022-07-26T04:18:51.604608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idx, train_Valx = train_test_split(list(labels.index), test_size=0.33, random_state=2021)\ntest_idx = list(sample_submission.index)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:51.607433Z","iopub.execute_input":"2022-07-26T04:18:51.608054Z","iopub.status.idle":"2022-07-26T04:18:51.850931Z","shell.execute_reply.started":"2022-07-26T04:18:51.608011Z","shell.execute_reply":"2022-07-26T04:18:51.849599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = DataGenerator('/kaggle/input/g2net-gravitational-wave-detection/train/', train_idx, labels, 64)\nval_generator = DataGenerator('/kaggle/input/g2net-gravitational-wave-detection/train/', train_Valx, labels, 64)\ntest_generator = DataGenerator('/kaggle/input/g2net-gravitational-wave-detection/test/', test_idx, sample_submission, 64)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:51.852398Z","iopub.execute_input":"2022-07-26T04:18:51.852893Z","iopub.status.idle":"2022-07-26T04:18:51.858055Z","shell.execute_reply.started":"2022-07-26T04:18:51.852836Z","shell.execute_reply":"2022-07-26T04:18:51.856910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv1D(64, input_shape=(3, 4096,), kernel_size=3, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:51.859809Z","iopub.execute_input":"2022-07-26T04:18:51.860353Z","iopub.status.idle":"2022-07-26T04:18:52.071579Z","shell.execute_reply.started":"2022-07-26T04:18:51.860307Z","shell.execute_reply":"2022-07-26T04:18:52.069824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = Adam(lr=2e-4),loss='binary_crossentropy',metrics=['acc'])","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:52.073010Z","iopub.execute_input":"2022-07-26T04:18:52.073393Z","iopub.status.idle":"2022-07-26T04:18:52.087385Z","shell.execute_reply.started":"2022-07-26T04:18:52.073353Z","shell.execute_reply":"2022-07-26T04:18:52.086437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:52.088845Z","iopub.execute_input":"2022-07-26T04:18:52.089758Z","iopub.status.idle":"2022-07-26T04:18:52.095519Z","shell.execute_reply.started":"2022-07-26T04:18:52.089697Z","shell.execute_reply":"2022-07-26T04:18:52.094339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator, validation_data=val_generator, epochs = 1, workers=4)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:18:52.098458Z","iopub.execute_input":"2022-07-26T04:18:52.099127Z","iopub.status.idle":"2022-07-26T04:39:09.444545Z","shell.execute_reply.started":"2022-07-26T04:18:52.099062Z","shell.execute_reply":"2022-07-26T04:39:09.440865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = model.predict_generator(test_generator, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T04:39:09.449687Z","iopub.execute_input":"2022-07-26T04:39:09.450942Z","iopub.status.idle":"2022-07-26T05:13:31.852384Z","shell.execute_reply.started":"2022-07-26T04:39:09.450864Z","shell.execute_reply":"2022-07-26T05:13:31.850041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission['target'] = predict[:len(sample_submission)]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T05:13:31.856291Z","iopub.execute_input":"2022-07-26T05:13:31.858525Z","iopub.status.idle":"2022-07-26T05:13:31.892831Z","shell.execute_reply.started":"2022-07-26T05:13:31.858442Z","shell.execute_reply":"2022-07-26T05:13:31.891850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T05:13:31.895334Z","iopub.execute_input":"2022-07-26T05:13:31.896290Z","iopub.status.idle":"2022-07-26T05:13:32.355685Z","shell.execute_reply.started":"2022-07-26T05:13:31.896252Z","shell.execute_reply":"2022-07-26T05:13:32.354025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission = pd.read_csv(\"./submission.csv\")\nmy_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-26T05:16:01.465368Z","iopub.execute_input":"2022-07-26T05:16:01.466352Z","iopub.status.idle":"2022-07-26T05:16:01.803868Z","shell.execute_reply.started":"2022-07-26T05:16:01.466295Z","shell.execute_reply":"2022-07-26T05:16:01.802276Z"},"trusted":true},"execution_count":null,"outputs":[]}]}