{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet\n!pip install nnAudio","metadata":{"execution":{"iopub.status.busy":"2021-08-17T19:32:09.173369Z","iopub.execute_input":"2021-08-17T19:32:09.173855Z","iopub.status.idle":"2021-08-17T19:32:28.877348Z","shell.execute_reply.started":"2021-08-17T19:32:09.173821Z","shell.execute_reply":"2021-08-17T19:32:28.876107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport scipy.signal as sig\nfrom scipy.signal import butter,lfilter\nfrom scipy.fft import fft,ifft\nimport pandas as pd\nimport os\nimport glob\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.wrappers.scikit_learn import KerasClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import SVC\nfrom sklearn.calibration import CalibratedClassifierCV\nimport datetime\nimport time\n#from tensorflow.keras.applications import EfficientNetB0\nfrom efficientnet.keras import EfficientNetB0\nimport torch\nfrom tensorflow.keras.utils import Sequence\nimport math\nfrom random import shuffle\nimport tensorflow.keras.layers as L\nimport pywt","metadata":{"execution":{"iopub.status.busy":"2021-08-17T20:07:01.762416Z","iopub.execute_input":"2021-08-17T20:07:01.762789Z","iopub.status.idle":"2021-08-17T20:07:01.772051Z","shell.execute_reply.started":"2021-08-17T20:07:01.762756Z","shell.execute_reply":"2021-08-17T20:07:01.770853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\nsample_submission = pd.read_csv('../input/g2net-gravitational-wave-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-17T20:07:03.042163Z","iopub.execute_input":"2021-08-17T20:07:03.042563Z","iopub.status.idle":"2021-08-17T20:07:03.60651Z","shell.execute_reply.started":"2021-08-17T20:07:03.04253Z","shell.execute_reply":"2021-08-17T20:07:03.603228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fs = 2048\nsteps_per_second = 32\nupsamp = fs//steps_per_second\nmin_width = max((0.1,steps_per_second/500))\nmax_width = steps_per_second/20\nwidths = np.arange(min_width, max_width, (max_width-min_width)/64)\nn_samples = 100000\n\ndef get_signal_id(i):\n    return targets.iloc[i]['id']\n\ndef get_path(id_no,is_train=True):\n    file_path = \"../input/g2net-gravitational-wave-detection/{}/{}/{}/{}/{}.npy\"\n    if is_train:\n        path = file_path.format('train',id_no[0],id_no[1],id_no[2],id_no)\n    else:\n        path = file_path.format('test',id_no[0],id_no[1],id_no[2],id_no)\n    return path    \n\ndef get_signal(id_no,is_train=True):\n    file = get_path(id_no,is_train)\n    signal = np.load(file)\n    return signal\n\ndef get_signals(id_no,is_train=True):\n    file = get_path(id_no,is_train)\n    signal = np.load(file)\n    return np.hstack(signal)\n\ndef get_filtered_scaled_signals(id_no,is_train=True):\n    signal = get_signal(id_no,is_train)   \n    return np.hstack(filter_scale_signals(signal))\n\ndef butter_bandpass(lowcut, highcut, fs, order=5):\n    nyq = 0.5 * fs\n    low = lowcut / nyq\n    high = highcut / nyq\n    b, a = butter(order, [low, high], btype='band')\n    return b, a\n\ndef butter_bandpass_filter(data, lowcut, highcut, fs, order=5):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y\n\ndef filter_scale_signal(signal):\n    tmp = butter_bandpass_filter(signal, 20, 500, fs, order=5)\n    tmp = 2*(tmp - min(tmp))/(max(tmp)-min(tmp))-1\n    return tmp\n\ndef filter_scale_signals(signal):\n    return [filter_scale_signal(s) for s in signal]\n\ndef get_scalogram(id_no,is_train):    \n    sigs = get_filtered_scaled_signals(id_no,is_train)\n    cwts = sig.cwt(sigs[0:-1:upsamp], sig.ricker, np.arange(1,65))\n    #cwts,freqs = pywt.cwt(sigs[:-1:upsamp],np.arange(1,65),'morl',sampling_period=1/2048)\n    return cwts.reshape((cwts.shape[0],cwts.shape[1],1))#out_mat = np.zeros((cwts.shape[0],cwts.shape[1]))\n\ndef get_model(shape):\n    model = tf.keras.Sequential([\n        L.InputLayer(input_shape=(shape[0],shape[1],1)),\n        L.Conv2D(3,3,activation='relu',padding='same'),\n        EfficientNetB0(include_top=False,input_shape=(),weights='imagenet'),\n        L.GlobalAveragePooling2D(),\n        L.Dense(32,activation='relu'),\n        L.Dense(1, activation='sigmoid')])\n\n    model.summary()\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n              loss='binary_crossentropy', metrics=[tf.keras.metrics.AUC()])\n    return model\n\nfrom nnAudio.Spectrogram import CQT1992v2\ndef get_spectrogram(idx,is_train,transform=CQT1992v2(sr=2048, fmin=20, fmax=1024, hop_length=64)): # in order to use efficientnet we need 3 dimension images\n    waves = np.load(get_path(idx,is_train))\n    waves = np.hstack(waves)\n    waves = waves / np.max(waves)\n    waves = torch.from_numpy(waves).float()\n    image = transform(waves)\n    image = np.array(image)\n    image = np.transpose(image,(1,2,0))\n    #image = np.array(get_scalogram(idx,is_train))\n    return image\n\nclass Dataset(Sequence):\n    def __init__(self,idx,y=None,batch_size=256,shuffle=True):\n        self.idx = idx\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        if y is not None:\n            self.is_train=True\n        else:\n            self.is_train=False\n        self.y = y\n    def __len__(self):\n        return math.ceil(len(self.idx)/self.batch_size)\n    def __getitem__(self,ids):\n        batch_ids = self.idx[ids * self.batch_size:(ids + 1) * self.batch_size]\n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n            \n        list_x = np.array([get_spectrogram(x,self.is_train) for x in batch_ids])\n        batch_X = np.stack(list_x)\n        if self.is_train:\n            return batch_X, batch_y\n        else:\n            return batch_X\n    \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.idx, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:05.063111Z","iopub.execute_input":"2021-08-17T22:54:05.063652Z","iopub.status.idle":"2021-08-17T22:54:05.119372Z","shell.execute_reply.started":"2021-08-17T22:54:05.063609Z","shell.execute_reply":"2021-08-17T22:54:05.11798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, t, Sxx = sig.spectrogram(get_filtered_scaled_signals(get_signal_id(0)), fs)\nplt.pcolormesh(t, f, Sxx, shading='gouraud')\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\nplt.ylim(1,100)\nplt.show()\nSxx.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:06.05949Z","iopub.execute_input":"2021-08-17T22:54:06.059891Z","iopub.status.idle":"2021-08-17T22:54:06.261722Z","shell.execute_reply.started":"2021-08-17T22:54:06.05986Z","shell.execute_reply":"2021-08-17T22:54:06.260488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(get_signals(get_signal_id(0)))","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:07.007874Z","iopub.execute_input":"2021-08-17T22:54:07.008284Z","iopub.status.idle":"2021-08-17T22:54:07.216341Z","shell.execute_reply.started":"2021-08-17T22:54:07.008223Z","shell.execute_reply":"2021-08-17T22:54:07.214997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(get_filtered_scaled_signals(get_signal_id(0)))","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:07.462794Z","iopub.execute_input":"2021-08-17T22:54:07.463156Z","iopub.status.idle":"2021-08-17T22:54:07.724541Z","shell.execute_reply.started":"2021-08-17T22:54:07.463123Z","shell.execute_reply":"2021-08-17T22:54:07.723424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(filter_scale_signal(get_signals(get_signal_id(0))))","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:07.726499Z","iopub.execute_input":"2021-08-17T22:54:07.727006Z","iopub.status.idle":"2021-08-17T22:54:07.950862Z","shell.execute_reply.started":"2021-08-17T22:54:07.726948Z","shell.execute_reply":"2021-08-17T22:54:07.949415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cwtmatr = get_scalogram(get_signal_id(0),True)\nplt.imshow(cwtmatr,aspect='auto')\nplt.show()\n\ncwtmatr.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:07.953479Z","iopub.execute_input":"2021-08-17T22:54:07.953901Z","iopub.status.idle":"2021-08-17T22:54:08.175147Z","shell.execute_reply.started":"2021-08-17T22:54:07.953858Z","shell.execute_reply":"2021-08-17T22:54:08.173783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cwtmatr = get_scalogram(get_signal_id(1),True)\nplt.imshow(cwtmatr,aspect='auto')\nplt.show()\n\ncwtmatr.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:08.177361Z","iopub.execute_input":"2021-08-17T22:54:08.177817Z","iopub.status.idle":"2021-08-17T22:54:08.401955Z","shell.execute_reply.started":"2021-08-17T22:54:08.177771Z","shell.execute_reply":"2021-08-17T22:54:08.400619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coefs,freqs = pywt.cwt(get_signals(get_signal_id(0))[:-1:64],np.arange(1,65),'morl',sampling_period=1/2048)\nfig = plt.figure()\nax = fig.gca()\nplt.imshow(coefs,aspect='auto')\nax.set_yticklabels(freqs)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:09.063732Z","iopub.execute_input":"2021-08-17T22:54:09.064158Z","iopub.status.idle":"2021-08-17T22:54:09.292732Z","shell.execute_reply.started":"2021-08-17T22:54:09.064124Z","shell.execute_reply":"2021-08-17T22:54:09.291401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coefs,freqs = pywt.cwt(get_filtered_scaled_signals(get_signal_id(0))[:-1:64],np.arange(1,65),'morl',sampling_period=1/2048)\nfig = plt.figure()\nax = fig.gca()\nplt.imshow(coefs,aspect='auto')\nax.set_yticklabels(freqs)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:10.337731Z","iopub.execute_input":"2021-08-17T22:54:10.338167Z","iopub.status.idle":"2021-08-17T22:54:10.649875Z","shell.execute_reply.started":"2021-08-17T22:54:10.338135Z","shell.execute_reply":"2021-08-17T22:54:10.648308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spect = get_spectrogram(get_signal_id(0),True)\nplt.imshow(spect)\nspect.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:11.601465Z","iopub.execute_input":"2021-08-17T22:54:11.601842Z","iopub.status.idle":"2021-08-17T22:54:11.835398Z","shell.execute_reply.started":"2021-08-17T22:54:11.601811Z","shell.execute_reply":"2021-08-17T22:54:11.833962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(np.log(abs(fft(get_filtered_scaled_signals(get_signal_id(0))))))\nplt.xlim(15,550)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:13.330829Z","iopub.execute_input":"2021-08-17T22:54:13.33121Z","iopub.status.idle":"2021-08-17T22:54:13.49618Z","shell.execute_reply.started":"2021-08-17T22:54:13.331177Z","shell.execute_reply":"2021-08-17T22:54:13.494801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(np.log(abs(fft(get_filtered_scaled_signals(get_signal_id(1))))))\nplt.xlim(15,550)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:14.383235Z","iopub.execute_input":"2021-08-17T22:54:14.383657Z","iopub.status.idle":"2021-08-17T22:54:14.541645Z","shell.execute_reply.started":"2021-08-17T22:54:14.383625Z","shell.execute_reply":"2021-08-17T22:54:14.540255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = targets#.sample(20000)\ntrain_idx =  samples['id'].values\ny = samples['target'].values\ntest_idx = sample_submission['id'].values\nx_train,x_valid,y_train,y_valid = train_test_split(train_idx,y,test_size=0.05,random_state=42,stratify=y)\n\ntrain_dataset = Dataset(x_train,y_train)\nvalid_dataset = Dataset(x_valid,y_valid)\ntest_dataset = Dataset(test_idx)\n\nmodel = get_model((69,193))","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:25.450524Z","iopub.execute_input":"2021-08-17T22:54:25.451099Z","iopub.status.idle":"2021-08-17T22:54:29.32313Z","shell.execute_reply.started":"2021-08-17T22:54:25.45105Z","shell.execute_reply":"2021-08-17T22:54:29.322088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_dataset,\n          epochs=1,\n          validation_data=valid_dataset)","metadata":{"execution":{"iopub.status.busy":"2021-08-17T22:54:31.389676Z","iopub.execute_input":"2021-08-17T22:54:31.390107Z","iopub.status.idle":"2021-08-18T00:51:52.131302Z","shell.execute_reply.started":"2021-08-17T22:54:31.390067Z","shell.execute_reply":"2021-08-18T00:51:52.128662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2021-08-18T00:51:52.137298Z","iopub.execute_input":"2021-08-18T00:51:52.137624Z","iopub.status.idle":"2021-08-18T01:33:39.252175Z","shell.execute_reply.started":"2021-08-18T00:51:52.137589Z","shell.execute_reply":"2021-08-18T01:33:39.246336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'id':sample_submission['id'],'target':preds.reshape(-1)})\ndf.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-18T01:33:39.263113Z","iopub.execute_input":"2021-08-18T01:33:39.263784Z","iopub.status.idle":"2021-08-18T01:33:39.957042Z","shell.execute_reply.started":"2021-08-18T01:33:39.263738Z","shell.execute_reply":"2021-08-18T01:33:39.955782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}