{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Going to perform data preparation and EDA","metadata":{}},{"cell_type":"markdown","source":"### Importing training labels file","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\ndf_train_label=pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\ndf_train_label.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:15:48.159736Z","iopub.execute_input":"2021-08-09T11:15:48.16018Z","iopub.status.idle":"2021-08-09T11:15:48.51829Z","shell.execute_reply.started":"2021-08-09T11:15:48.160147Z","shell.execute_reply":"2021-08-09T11:15:48.516878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:15:48.52066Z","iopub.execute_input":"2021-08-09T11:15:48.521177Z","iopub.status.idle":"2021-08-09T11:15:48.529099Z","shell.execute_reply.started":"2021-08-09T11:15:48.521131Z","shell.execute_reply":"2021-08-09T11:15:48.527901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:15:48.532171Z","iopub.execute_input":"2021-08-09T11:15:48.533003Z","iopub.status.idle":"2021-08-09T11:15:48.541867Z","shell.execute_reply.started":"2021-08-09T11:15:48.532956Z","shell.execute_reply":"2021-08-09T11:15:48.540425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Importing tha path to the files","metadata":{}},{"cell_type":"code","source":"# path of the files\npaths_files = glob(\"../input/g2net-gravitational-wave-detection/train/*/*/*/*\")\n#paths_files","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:15:48.544421Z","iopub.execute_input":"2021-08-09T11:15:48.54537Z","iopub.status.idle":"2021-08-09T11:16:30.372078Z","shell.execute_reply.started":"2021-08-09T11:15:48.5453Z","shell.execute_reply":"2021-08-09T11:16:30.370956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(paths_files)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.373681Z","iopub.execute_input":"2021-08-09T11:16:30.374112Z","iopub.status.idle":"2021-08-09T11:16:30.382612Z","shell.execute_reply.started":"2021-08-09T11:16:30.374069Z","shell.execute_reply":"2021-08-09T11:16:30.381164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are about 560000 **.npy** files in the dataframe. Now we look into a particular .npy file as shown below.","metadata":{}},{"cell_type":"code","source":"# Loading the first .npy data\ndata=np.load(paths_files[0])\ndata","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.384864Z","iopub.execute_input":"2021-08-09T11:16:30.38545Z","iopub.status.idle":"2021-08-09T11:16:30.404929Z","shell.execute_reply.started":"2021-08-09T11:16:30.385404Z","shell.execute_reply":"2021-08-09T11:16:30.403798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.406468Z","iopub.execute_input":"2021-08-09T11:16:30.407128Z","iopub.status.idle":"2021-08-09T11:16:30.414449Z","shell.execute_reply.started":"2021-08-09T11:16:30.40707Z","shell.execute_reply":"2021-08-09T11:16:30.413123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From the above observation we can conclude the following,\n1. The sampling rate is 2048 Hz, which means that for each second 2048 samples are given. This fact is already given in the dataset description.\n2. Three rows in **data** variable refer to the 3 sites mentioned in the description of data, and they are: LIGO Hanford (SITE1), LIGO Livingston (SITE2), Virgo (SITE3).\n3. In the **data** variable there are $4096=2086\\times 2$ columns. It refers to the total samples generated in the span of 2 seconds.\n","metadata":{}},{"cell_type":"code","source":"print(np.min(data),np.max(data))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.419138Z","iopub.execute_input":"2021-08-09T11:16:30.420092Z","iopub.status.idle":"2021-08-09T11:16:30.427836Z","shell.execute_reply.started":"2021-08-09T11:16:30.420031Z","shell.execute_reply":"2021-08-09T11:16:30.426369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Looking is there is any missing value\ndf_train_label.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.431163Z","iopub.execute_input":"2021-08-09T11:16:30.432175Z","iopub.status.idle":"2021-08-09T11:16:30.49369Z","shell.execute_reply.started":"2021-08-09T11:16:30.432129Z","shell.execute_reply":"2021-08-09T11:16:30.49226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label['target'].hist()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.495697Z","iopub.execute_input":"2021-08-09T11:16:30.496339Z","iopub.status.idle":"2021-08-09T11:16:30.802358Z","shell.execute_reply.started":"2021-08-09T11:16:30.496292Z","shell.execute_reply":"2021-08-09T11:16:30.800892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.804488Z","iopub.execute_input":"2021-08-09T11:16:30.805055Z","iopub.status.idle":"2021-08-09T11:16:30.824366Z","shell.execute_reply.started":"2021-08-09T11:16:30.804977Z","shell.execute_reply":"2021-08-09T11:16:30.823046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost balanced data.\n\n#### Motivated by the compact dataset representation in the kaggle notebook given [here](https://www.kaggle.com/rawaaelghali/g2net-gravitational-starter-eda) we also build similar compact dataframe as shown below.","metadata":{}},{"cell_type":"code","source":"ids=[]\nfor filext in paths_files:\n    ids.append(filext[filext.rindex('/')+1:\\\n                              len(filext)].replace('.npy',''))\n    \n# data frame containing paths and ids of .npy files \npath_df = pd.DataFrame({\"id\":ids,\"path\":paths_files})\npath_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:30.826223Z","iopub.execute_input":"2021-08-09T11:16:30.826731Z","iopub.status.idle":"2021-08-09T11:16:31.51205Z","shell.execute_reply.started":"2021-08-09T11:16:30.826685Z","shell.execute_reply":"2021-08-09T11:16:31.510723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:31.513871Z","iopub.execute_input":"2021-08-09T11:16:31.514357Z","iopub.status.idle":"2021-08-09T11:16:31.522545Z","shell.execute_reply.started":"2021-08-09T11:16:31.514311Z","shell.execute_reply":"2021-08-09T11:16:31.521184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We know do a join both of the dataframes into a resulting dataframe **df**","metadata":{}},{"cell_type":"code","source":"df=pd.merge(path_df,df_train_label,on='id')\ndel path_df, df_train_label;\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:31.524472Z","iopub.execute_input":"2021-08-09T11:16:31.524963Z","iopub.status.idle":"2021-08-09T11:16:32.138132Z","shell.execute_reply.started":"2021-08-09T11:16:31.524917Z","shell.execute_reply":"2021-08-09T11:16:32.136568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:32.140262Z","iopub.execute_input":"2021-08-09T11:16:32.140765Z","iopub.status.idle":"2021-08-09T11:16:32.149531Z","shell.execute_reply.started":"2021-08-09T11:16:32.14072Z","shell.execute_reply":"2021-08-09T11:16:32.14799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['target']==1]['path'][0]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:32.151391Z","iopub.execute_input":"2021-08-09T11:16:32.152029Z","iopub.status.idle":"2021-08-09T11:16:32.226847Z","shell.execute_reply.started":"2021-08-09T11:16:32.151983Z","shell.execute_reply":"2021-08-09T11:16:32.225799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualizing a particular .npy file where target=0 and target=1. ","metadata":{}},{"cell_type":"code","source":"for i in range(0,data.shape[0]):\n    plt.plot(np.arange(0, data.shape[1], 1),data[i,:])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:32.228643Z","iopub.execute_input":"2021-08-09T11:16:32.229113Z","iopub.status.idle":"2021-08-09T11:16:32.705389Z","shell.execute_reply.started":"2021-08-09T11:16:32.229068Z","shell.execute_reply":"2021-08-09T11:16:32.704269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### First we look into the case when target=1","metadata":{}},{"cell_type":"code","source":"data1=np.load(df[df['target']==1]['path'].iloc[0])\ndata1","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:32.70695Z","iopub.execute_input":"2021-08-09T11:16:32.707447Z","iopub.status.idle":"2021-08-09T11:16:32.749882Z","shell.execute_reply.started":"2021-08-09T11:16:32.707401Z","shell.execute_reply":"2021-08-09T11:16:32.748489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,data1.shape[0]): \n    plt.figure(figsize=(14,2))\n    plt.plot(np.arange(0, data1.shape[1], 1),data1[i,:])\n    # naming the x axis\n    plt.xlabel('sample')\n    # naming the y axis\n    plt.ylabel('output')\n    # naming the title\n    plt.title('SITE'+str(i+1)+'(target=1)')\n    plt.xlim(0,4096)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:32.751827Z","iopub.execute_input":"2021-08-09T11:16:32.75229Z","iopub.status.idle":"2021-08-09T11:16:33.277761Z","shell.execute_reply.started":"2021-08-09T11:16:32.752244Z","shell.execute_reply":"2021-08-09T11:16:33.276641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We now look into target=0","metadata":{}},{"cell_type":"code","source":"data0=np.load(df[df['target']==0]['path'].iloc[0])\ndata0","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:33.279428Z","iopub.execute_input":"2021-08-09T11:16:33.279926Z","iopub.status.idle":"2021-08-09T11:16:33.331717Z","shell.execute_reply.started":"2021-08-09T11:16:33.27988Z","shell.execute_reply":"2021-08-09T11:16:33.33059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,data0.shape[0]): \n    plt.figure(figsize=(14,2))\n    plt.plot(np.arange(0, data0.shape[1], 1),data0[i,:])\n    # naming the x axis\n    plt.xlabel('sample')\n    # naming the y axis\n    plt.ylabel('output')\n    # naming the title\n    plt.title('SITE'+str(i+1)+'(target=0)')\n    plt.xlim(0,4096)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:33.333768Z","iopub.execute_input":"2021-08-09T11:16:33.334239Z","iopub.status.idle":"2021-08-09T11:16:33.992446Z","shell.execute_reply.started":"2021-08-09T11:16:33.334193Z","shell.execute_reply":"2021-08-09T11:16:33.991262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['target']==0]['path'].iloc[0]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:33.994208Z","iopub.execute_input":"2021-08-09T11:16:33.994881Z","iopub.status.idle":"2021-08-09T11:16:34.035161Z","shell.execute_reply.started":"2021-08-09T11:16:33.994833Z","shell.execute_reply":"2021-08-09T11:16:34.033737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data1[0,:])","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:34.040111Z","iopub.execute_input":"2021-08-09T11:16:34.040643Z","iopub.status.idle":"2021-08-09T11:16:34.495675Z","shell.execute_reply.started":"2021-08-09T11:16:34.040583Z","shell.execute_reply":"2021-08-09T11:16:34.494408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 2, sharex=True, figsize=(14,12))\nfig.suptitle('Distribution plots')\nfor i in range(0,data1.shape[0]):\n    sns.histplot(ax=axes[i, 0], data=data1[i,:])\n    axes[i,0].set_title('SITE'+str(i+1)+'(target=1)')\n    sns.histplot(ax=axes[i, 1], data=data0[i,:])\n    axes[i,1].set_title('SITE'+str(i+1)+'(target=0)')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:34.497734Z","iopub.execute_input":"2021-08-09T11:16:34.498187Z","iopub.status.idle":"2021-08-09T11:16:36.353361Z","shell.execute_reply.started":"2021-08-09T11:16:34.49814Z","shell.execute_reply":"2021-08-09T11:16:36.352315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It appears that target=1 at SITE1 has higher spread, and target=0 has higher spread at SITE3","metadata":{}},{"cell_type":"markdown","source":"#### Creating Training and validation set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train, df_val= train_test_split(df, test_size=0.2, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:36.355086Z","iopub.execute_input":"2021-08-09T11:16:36.355668Z","iopub.status.idle":"2021-08-09T11:16:36.726832Z","shell.execute_reply.started":"2021-08-09T11:16:36.35558Z","shell.execute_reply":"2021-08-09T11:16:36.725727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:16:36.728287Z","iopub.execute_input":"2021-08-09T11:16:36.728715Z","iopub.status.idle":"2021-08-09T11:16:36.740598Z","shell.execute_reply.started":"2021-08-09T11:16:36.728673Z","shell.execute_reply":"2021-08-09T11:16:36.739307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:08.111229Z","iopub.execute_input":"2021-08-09T11:18:08.111668Z","iopub.status.idle":"2021-08-09T11:18:08.128463Z","shell.execute_reply.started":"2021-08-09T11:18:08.111625Z","shell.execute_reply":"2021-08-09T11:18:08.127002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now in inorder to build custom data generator, do refer to this article [ref.(1)](https://towardsdatascience.com/keras-data-generators-and-how-to-use-them-b69129ed779c) it is extremely useful.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Conv1D, MaxPool1D, BatchNormalization\nfrom tensorflow.keras.optimizers import RMSprop,Adam\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:13.759824Z","iopub.execute_input":"2021-08-09T11:18:13.760195Z","iopub.status.idle":"2021-08-09T11:18:18.93482Z","shell.execute_reply.started":"2021-08-09T11:18:13.760162Z","shell.execute_reply":"2021-08-09T11:18:18.933657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Directly training our model on the **.npy** files takes a lot of time because loading the data takes a lot of time than performing the ML computations. Mainly the ML computation are done on a GPU, and loading the data task is done by CPU. Former data pipelines made the GPU wait for the CPU to load the data, leading to performance issues [ref.(2)](https://cs230.stanford.edu/blog/datapipeline/). Therefore, the **tf.data** API enables you to build complex input pipelines from simple, reusable pieces [ref.(3)](https://www.tensorflow.org/guide/data). But this (**tf.data**) API lack the feature of reading **.npy** files which doesn't fit in the memory. However, I found a solution in the stackover flow, and link of the solution is given [here](https://stackoverflow.com/questions/48889482/feeding-npy-numpy-files-into-tensorflow-data-pipeline).","metadata":{}},{"cell_type":"markdown","source":"#### It is actually possible to read directly NPY files with TensorFlow instead of TFRecords. The key pieces are [**tf.data.FixedLengthRecordDataset**](https://www.tensorflow.org/api_docs/python/tf/data/FixedLengthRecordDataset) and [**tf.io.decode_raw**](https://www.tensorflow.org/api_docs/python/tf/io/decode_raw), along with a look at the documentation of the [**.npy**](https://numpy.org/devdocs/reference/generated/numpy.lib.format.html) format. For simplicity, let's suppose that a **float32** **.npy** file containing an array with shape (N, K) is given, and you know the number of features K beforehand, as well as the fact that it is a *float32* array. An **.npy** file is just a binary file with a small header and followed by the raw array data (object arrays are different, but we're considering numbers now). In short, you can find the size of this header with a function like this:","metadata":{}},{"cell_type":"code","source":"def npy_header_offset(npy_path):\n    with open(str(npy_path), 'rb') as f:\n        if f.read(6) != b'\\x93NUMPY':\n            raise ValueError('Invalid NPY file.')\n        version_major, version_minor = f.read(2)\n        if version_major == 1:\n            header_len_size = 2\n        elif version_major == 2:\n            header_len_size = 4\n        else:\n            raise ValueError('Unknown NPY file version {}.{}.'.format(version_major, version_minor))\n        header_len = sum(b << (8 * i) for i, b in enumerate(f.read(header_len_size)))\n        header = f.read(header_len)\n        if not header.endswith(b'\\n'):\n            raise ValueError('Invalid NPY file.')\n        return f.tell()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:27.160827Z","iopub.execute_input":"2021-08-09T11:18:27.161217Z","iopub.status.idle":"2021-08-09T11:18:27.169818Z","shell.execute_reply.started":"2021-08-09T11:18:27.161184Z","shell.execute_reply":"2021-08-09T11:18:27.168443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"header_size = npy_header_offset(df['path'].iloc[0])\nheader_size","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:37.194426Z","iopub.execute_input":"2021-08-09T11:18:37.194856Z","iopub.status.idle":"2021-08-09T11:18:37.21174Z","shell.execute_reply.started":"2021-08-09T11:18:37.194822Z","shell.execute_reply":"2021-08-09T11:18:37.210682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_length = os.path.getsize(df['path'].iloc[0])\nfile_length","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:40.878627Z","iopub.execute_input":"2021-08-09T11:18:40.879026Z","iopub.status.idle":"2021-08-09T11:18:40.886949Z","shell.execute_reply.started":"2021-08-09T11:18:40.878995Z","shell.execute_reply":"2021-08-09T11:18:40.885707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_length-header_size","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:46.416462Z","iopub.execute_input":"2021-08-09T11:18:46.416929Z","iopub.status.idle":"2021-08-09T11:18:46.424591Z","shell.execute_reply.started":"2021-08-09T11:18:46.416895Z","shell.execute_reply":"2021-08-09T11:18:46.423142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"3*4096*tf.float64.size","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:18:52.036055Z","iopub.execute_input":"2021-08-09T11:18:52.036438Z","iopub.status.idle":"2021-08-09T11:18:52.0451Z","shell.execute_reply.started":"2021-08-09T11:18:52.036405Z","shell.execute_reply":"2021-08-09T11:18:52.043615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Documentation regarding tf.data.FixedLengthRecordDataset is given [here](https://www.tensorflow.org/api_docs/python/tf/data/FixedLengthRecordDataset)","metadata":{}},{"cell_type":"code","source":"tf_data_train=tf.data.FixedLengthRecordDataset( df_train['path'], 3*4096*tf.float64.size,\\\n                                         header_bytes=header_size, num_parallel_reads=4)\ntf_data_val=tf.data.FixedLengthRecordDataset( df_val['path'], 3*4096*tf.float64.size,\\\n                                         header_bytes=header_size, num_parallel_reads=4)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:19:43.135627Z","iopub.execute_input":"2021-08-09T11:19:43.135991Z","iopub.status.idle":"2021-08-09T11:19:45.405941Z","shell.execute_reply.started":"2021-08-09T11:19:43.135958Z","shell.execute_reply":"2021-08-09T11:19:45.404851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_data_train","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:19:52.965156Z","iopub.execute_input":"2021-08-09T11:19:52.965581Z","iopub.status.idle":"2021-08-09T11:19:52.978055Z","shell.execute_reply.started":"2021-08-09T11:19:52.965544Z","shell.execute_reply":"2021-08-09T11:19:52.976745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_data_train = tf_data_train.map(lambda s: tf.reshape(\\\n                                                       tf.io.decode_raw(s, tf.float64),\\\n                                                       (3,4096)))\ntf_data_val = tf_data_val.map(lambda s: tf.reshape(\\\n                                                       tf.io.decode_raw(s, tf.float64),\\\n                                                       (3,4096)))\ntf_data_train","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:20:43.039095Z","iopub.execute_input":"2021-08-09T11:20:43.039463Z","iopub.status.idle":"2021-08-09T11:20:43.113201Z","shell.execute_reply.started":"2021-08-09T11:20:43.03943Z","shell.execute_reply":"2021-08-09T11:20:43.111429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tf_data_train.take(3):\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:21:01.785275Z","iopub.execute_input":"2021-08-09T11:21:01.785637Z","iopub.status.idle":"2021-08-09T11:21:01.922008Z","shell.execute_reply.started":"2021-08-09T11:21:01.785606Z","shell.execute_reply":"2021-08-09T11:21:01.92072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tf_data_val.take(3):\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:21:19.187949Z","iopub.execute_input":"2021-08-09T11:21:19.188341Z","iopub.status.idle":"2021-08-09T11:21:19.234793Z","shell.execute_reply.started":"2021-08-09T11:21:19.188309Z","shell.execute_reply":"2021-08-09T11:21:19.233673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### I am loving it.\n\n### Now going to zip the target column with the tensorflow dataset","metadata":{}},{"cell_type":"code","source":"tf_data_train= tf.data.Dataset.zip((tf_data_train,\\\n                             tf.data.Dataset.from_tensor_slices(df_train['target']))) ","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:22:02.407284Z","iopub.execute_input":"2021-08-09T11:22:02.407659Z","iopub.status.idle":"2021-08-09T11:22:02.417809Z","shell.execute_reply.started":"2021-08-09T11:22:02.407626Z","shell.execute_reply":"2021-08-09T11:22:02.416589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nfor data, target in tf_data_train.take(3):\n    print(\"tf_data_train\")\n    print(data.numpy(),target.numpy())\n    print(\"df_train\")\n    print(np.load(df_train['path'].iloc[i]),df_train['target'].iloc[i])\n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:31:53.566707Z","iopub.execute_input":"2021-08-09T11:31:53.567148Z","iopub.status.idle":"2021-08-09T11:31:53.62909Z","shell.execute_reply.started":"2021-08-09T11:31:53.567114Z","shell.execute_reply":"2021-08-09T11:31:53.628009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_data_val= tf.data.Dataset.zip((tf_data_val,\\\n                             tf.data.Dataset.from_tensor_slices(df_val['target']))) \nfor data, target in tf_data_val.take(3):\n    print(data.numpy(),target.numpy())","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:23:08.974539Z","iopub.execute_input":"2021-08-09T11:23:08.974924Z","iopub.status.idle":"2021-08-09T11:23:09.021036Z","shell.execute_reply.started":"2021-08-09T11:23:08.974892Z","shell.execute_reply":"2021-08-09T11:23:09.02001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = tf_data_train.batch(32).prefetch(buffer_size=64)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2021-08-09T11:57:47.304215Z","iopub.execute_input":"2021-08-09T11:57:47.304628Z","iopub.status.idle":"2021-08-09T11:57:47.316515Z","shell.execute_reply.started":"2021-08-09T11:57:47.304582Z","shell.execute_reply":"2021-08-09T11:57:47.314999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_data = tf_data_val.batch(32).prefetch(buffer_size=64)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:00:20.185669Z","iopub.execute_input":"2021-08-09T12:00:20.186068Z","iopub.status.idle":"2021-08-09T12:00:20.197964Z","shell.execute_reply.started":"2021-08-09T12:00:20.186019Z","shell.execute_reply":"2021-08-09T12:00:20.19642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nmodel = Sequential()\nmodel.add(Conv1D(64, input_shape=(3, 4096,), kernel_size=3, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:00:57.866638Z","iopub.execute_input":"2021-08-09T12:00:57.867236Z","iopub.status.idle":"2021-08-09T12:00:58.362497Z","shell.execute_reply.started":"2021-08-09T12:00:57.867183Z","shell.execute_reply":"2021-08-09T12:00:58.361379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = Adam(lr=2e-4),loss='binary_crossentropy',metrics=['AUC'])","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:01:03.59179Z","iopub.execute_input":"2021-08-09T12:01:03.592188Z","iopub.status.idle":"2021-08-09T12:01:03.612189Z","shell.execute_reply.started":"2021-08-09T12:01:03.592141Z","shell.execute_reply":"2021-08-09T12:01:03.610786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:01:06.363252Z","iopub.execute_input":"2021-08-09T12:01:06.363677Z","iopub.status.idle":"2021-08-09T12:01:06.377456Z","shell.execute_reply.started":"2021-08-09T12:01:06.363648Z","shell.execute_reply":"2021-08-09T12:01:06.375342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_data, validation_data=val_data, epochs = 2)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:01:40.67415Z","iopub.execute_input":"2021-08-09T12:01:40.67454Z","iopub.status.idle":"2021-08-09T12:18:03.489752Z","shell.execute_reply.started":"2021-08-09T12:01:40.674483Z","shell.execute_reply":"2021-08-09T12:18:03.488587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path of the files\ntest_files = glob(\"../input/g2net-gravitational-wave-detection/test/*/*/*/*\")\n#paths_files","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:18:34.231258Z","iopub.execute_input":"2021-08-09T12:18:34.23171Z","iopub.status.idle":"2021-08-09T12:19:15.370899Z","shell.execute_reply.started":"2021-08-09T12:18:34.231675Z","shell.execute_reply":"2021-08-09T12:19:15.369763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids=[]\nfor filext in test_files:\n    ids.append(filext[filext.rindex('/')+1:\\\n                              len(filext)].replace('.npy',''))\n    \n# data frame containing paths and ids of .npy files \ntest_df = pd.DataFrame({\"id\":ids,\"path\":test_files})\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:21:08.155402Z","iopub.execute_input":"2021-08-09T12:21:08.155877Z","iopub.status.idle":"2021-08-09T12:21:08.46142Z","shell.execute_reply.started":"2021-08-09T12:21:08.155844Z","shell.execute_reply":"2021-08-09T12:21:08.459991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_data_test=tf.data.FixedLengthRecordDataset( test_df['path'], 3*4096*tf.float64.size,\\\n                                         header_bytes=header_size, num_parallel_reads=4)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:22:56.103131Z","iopub.execute_input":"2021-08-09T12:22:56.103573Z","iopub.status.idle":"2021-08-09T12:22:56.143737Z","shell.execute_reply.started":"2021-08-09T12:22:56.103536Z","shell.execute_reply":"2021-08-09T12:22:56.142589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_data_test = tf_data_test.map(lambda s: tf.reshape(\\\n                                                       tf.io.decode_raw(s, tf.float64),\\\n                                                       (3,4096)))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:23:37.909266Z","iopub.execute_input":"2021-08-09T12:23:37.909712Z","iopub.status.idle":"2021-08-09T12:23:37.924542Z","shell.execute_reply.started":"2021-08-09T12:23:37.909681Z","shell.execute_reply":"2021-08-09T12:23:37.923337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = tf_data_test.batch(32).prefetch(buffer_size=64)\ny_pred=model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:26:22.217677Z","iopub.execute_input":"2021-08-09T12:26:22.218073Z","iopub.status.idle":"2021-08-09T12:29:42.040559Z","shell.execute_reply.started":"2021-08-09T12:26:22.218039Z","shell.execute_reply":"2021-08-09T12:29:42.039577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.flatten()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:33:12.297119Z","iopub.execute_input":"2021-08-09T12:33:12.297604Z","iopub.status.idle":"2021-08-09T12:33:12.305108Z","shell.execute_reply.started":"2021-08-09T12:33:12.297556Z","shell.execute_reply":"2021-08-09T12:33:12.303892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'Id': test_df.id, 'target': y_pred.flatten()})\noutput.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:33:20.841183Z","iopub.execute_input":"2021-08-09T12:33:20.841614Z","iopub.status.idle":"2021-08-09T12:33:20.86448Z","shell.execute_reply.started":"2021-08-09T12:33:20.841574Z","shell.execute_reply":"2021-08-09T12:33:20.863114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('./testing_submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:34:20.536789Z","iopub.execute_input":"2021-08-09T12:34:20.537182Z","iopub.status.idle":"2021-08-09T12:34:21.184285Z","shell.execute_reply.started":"2021-08-09T12:34:20.53715Z","shell.execute_reply":"2021-08-09T12:34:21.182909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}