{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries, Load Paths to `pd.DataFrame`","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pickle\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nid_2_path = dict()\nfor dirname, _, filenames in tqdm(os.walk('/kaggle/input/g2net-gravitational-wave-detection/train/'), total=4369, desc='Checking filepath for each ID...'):\n    for filename in filenames:\n        if not os.path.isdir(filename):\n            id_2_path[os.path.splitext(filename)[0]] = os.path.join(dirname, filename)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-30T18:48:12.435009Z","iopub.execute_input":"2021-06-30T18:48:12.435377Z","iopub.status.idle":"2021-06-30T18:49:56.669883Z","shell.execute_reply.started":"2021-06-30T18:48:12.435346Z","shell.execute_reply":"2021-06-30T18:49:56.668941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Dataset Paths\n\nTook help from [this Notebook](https://www.kaggle.com/samusram/g2net-mapping-id-to-file-path-eda?select=training_labels_with_paths.csv).","metadata":{}},{"cell_type":"code","source":"training_labels = pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\ntraining_labels['filepath'] = training_labels['id'].map(id_2_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:56.67178Z","iopub.execute_input":"2021-06-30T18:49:56.672122Z","iopub.status.idle":"2021-06-30T18:49:57.693707Z","shell.execute_reply.started":"2021-06-30T18:49:56.672091Z","shell.execute_reply":"2021-06-30T18:49:57.692625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.694958Z","iopub.execute_input":"2021-06-30T18:49:57.695298Z","iopub.status.idle":"2021-06-30T18:49:57.712867Z","shell.execute_reply.started":"2021-06-30T18:49:57.695267Z","shell.execute_reply":"2021-06-30T18:49:57.711953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.714166Z","iopub.execute_input":"2021-06-30T18:49:57.714461Z","iopub.status.idle":"2021-06-30T18:49:57.721048Z","shell.execute_reply.started":"2021-06-30T18:49:57.714433Z","shell.execute_reply":"2021-06-30T18:49:57.719901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels.to_csv(\"full_data.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T19:00:53.224425Z","iopub.execute_input":"2021-06-30T19:00:53.224907Z","iopub.status.idle":"2021-06-30T19:00:56.382162Z","shell.execute_reply.started":"2021-06-30T19:00:53.224865Z","shell.execute_reply":"2021-06-30T19:00:56.381187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define Property of Array Helper","metadata":{}},{"cell_type":"code","source":"def props(arr):\n    print(\"Shape :\",arr.shape,\"Maximum :\",arr.max(),\"Minimum :\",arr.min())","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.724397Z","iopub.execute_input":"2021-06-30T18:49:57.724871Z","iopub.status.idle":"2021-06-30T18:49:57.731488Z","shell.execute_reply.started":"2021-06-30T18:49:57.724802Z","shell.execute_reply":"2021-06-30T18:49:57.730392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for row_i, pos_filepath in enumerate(training_labels['filepath'].sample(10)):\n    pos_data = np.load(pos_filepath)\n    props(pos_data)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.733534Z","iopub.execute_input":"2021-06-30T18:49:57.734008Z","iopub.status.idle":"2021-06-30T18:49:57.841783Z","shell.execute_reply.started":"2021-06-30T18:49:57.733958Z","shell.execute_reply":"2021-06-30T18:49:57.840686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from copy import deepcopy\ndf_full = deepcopy(training_labels)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.843449Z","iopub.execute_input":"2021-06-30T18:49:57.843897Z","iopub.status.idle":"2021-06-30T18:49:57.872402Z","shell.execute_reply.started":"2021-06-30T18:49:57.843827Z","shell.execute_reply":"2021-06-30T18:49:57.871368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Sampling of Train Dataset\n\n### As initial data is massive!","metadata":{}},{"cell_type":"code","source":"df = df_full.sample(10000, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.874021Z","iopub.execute_input":"2021-06-30T18:49:57.874488Z","iopub.status.idle":"2021-06-30T18:49:57.970782Z","shell.execute_reply.started":"2021-06-30T18:49:57.874439Z","shell.execute_reply":"2021-06-30T18:49:57.969668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Perform `Train - Test` Split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain, test = train_test_split(df, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:57.972353Z","iopub.execute_input":"2021-06-30T18:49:57.972769Z","iopub.status.idle":"2021-06-30T18:49:59.107029Z","shell.execute_reply.started":"2021-06-30T18:49:57.972726Z","shell.execute_reply":"2021-06-30T18:49:59.106077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train), len(test))","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:59.108535Z","iopub.execute_input":"2021-06-30T18:49:59.108879Z","iopub.status.idle":"2021-06-30T18:49:59.113722Z","shell.execute_reply.started":"2021-06-30T18:49:59.108823Z","shell.execute_reply":"2021-06-30T18:49:59.112657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import LSTM\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:49:59.115351Z","iopub.execute_input":"2021-06-30T18:49:59.115794Z","iopub.status.idle":"2021-06-30T18:50:05.375396Z","shell.execute_reply.started":"2021-06-30T18:49:59.115748Z","shell.execute_reply":"2021-06-30T18:50:05.374484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fix `SEED` for Code Reproducibility","metadata":{}},{"cell_type":"code","source":"# fix random seed for reproducibility\nnp.random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:50:05.376824Z","iopub.execute_input":"2021-06-30T18:50:05.377429Z","iopub.status.idle":"2021-06-30T18:50:05.382449Z","shell.execute_reply.started":"2021-06-30T18:50:05.377384Z","shell.execute_reply":"2021-06-30T18:50:05.381114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# finished all my gpu else would go 4 this\nmodel = Sequential()\nmodel.add(LSTM(4096,input_shape=(3, 4096))) # 12288\nmodel.add(LSTM(1024,input_shape=(3, 4096))) # 12288\nmodel.add(LSTM(256,input_shape=(3, 4096))) # 12288\nmodel.add(LSTM(64,input_shape=(3, 4096))) # 12288\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\"\"\"\n\n\nmodel = Sequential()\nmodel.add(LSTM(100,input_shape=(3, 4096))) # 12288\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:55:06.704788Z","iopub.execute_input":"2021-06-30T18:55:06.70535Z","iopub.status.idle":"2021-06-30T18:55:06.962793Z","shell.execute_reply.started":"2021-06-30T18:55:06.705303Z","shell.execute_reply":"2021-06-30T18:55:06.9619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.layers","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:55:07.17346Z","iopub.execute_input":"2021-06-30T18:55:07.174006Z","iopub.status.idle":"2021-06-30T18:55:07.17888Z","shell.execute_reply.started":"2021-06-30T18:55:07.173958Z","shell.execute_reply":"2021-06-30T18:55:07.178024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:55:08.152275Z","iopub.execute_input":"2021-06-30T18:55:08.152797Z","iopub.status.idle":"2021-06-30T18:55:08.160376Z","shell.execute_reply.started":"2021-06-30T18:55:08.152752Z","shell.execute_reply":"2021-06-30T18:55:08.15925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:55:09.062549Z","iopub.execute_input":"2021-06-30T18:55:09.062926Z","iopub.status.idle":"2021-06-30T18:55:09.076109Z","shell.execute_reply.started":"2021-06-30T18:55:09.062891Z","shell.execute_reply":"2021-06-30T18:55:09.074972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Custom Generator\nTo perform Training, we must make our own `DataGenerator` by subclassing from `keras.utils.Sequence` class.","metadata":{}},{"cell_type":"code","source":"# https://towardsdatascience.com/implementing-custom-data-generators-in-keras-de56f013581c\nclass DataGenerator(keras.utils.Sequence):\n    def __init__(self, df, x_col=\"filepath\", y_col=\"target\", batch_size=32, num_classes=2, shuffle=True):\n        self.batch_size = batch_size\n        self.df = df\n        self.indices = self.df.index.tolist()\n        self.num_classes = num_classes\n        self.shuffle = shuffle\n        self.x_col = x_col\n        self.y_col = y_col\n        self.on_epoch_end()\n\n    def __len__(self):\n        return len(self.indices) // self.batch_size\n\n    def __getitem__(self, index):\n        index = self.index[index * self.batch_size:(index + 1) * self.batch_size]\n        batch = [self.indices[k] for k in index]\n        \n        X, y = self.__get_data(batch)\n        return X, y\n\n    def on_epoch_end(self):\n        self.index = np.arange(len(self.indices))\n        if self.shuffle == True:\n            np.random.shuffle(self.index)\n\n    def __get_data(self, batch):\n        length = len(batch)\n        X = np.zeros((length,3, 4096))\n        y = np.zeros(length)\n        \n        for i, id in enumerate(batch):\n            #print(id)\n            #print(df.loc[id])\n            X[i,] = np.load(self.df.loc[id][self.x_col])\n            y[i] = self.df.loc[id][self.y_col]\n\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:55:11.713367Z","iopub.execute_input":"2021-06-30T18:55:11.713701Z","iopub.status.idle":"2021-06-30T18:55:11.726261Z","shell.execute_reply.started":"2021-06-30T18:55:11.713672Z","shell.execute_reply":"2021-06-30T18:55:11.724885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_generator = DataGenerator(train)\nvalidation_generator = DataGenerator(test)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:56:49.507126Z","iopub.execute_input":"2021-06-30T18:56:49.50774Z","iopub.status.idle":"2021-06-30T18:56:49.514749Z","shell.execute_reply.started":"2021-06-30T18:56:49.507696Z","shell.execute_reply":"2021-06-30T18:56:49.513708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"# Train model on dataset\nhist = model.fit_generator(generator=training_generator,\n                    validation_data=validation_generator,\n                    use_multiprocessing=True,\n                    workers=6, epochs=50)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:56:50.552828Z","iopub.execute_input":"2021-06-30T18:56:50.553446Z","iopub.status.idle":"2021-06-30T18:58:23.054333Z","shell.execute_reply.started":"2021-06-30T18:56:50.5534Z","shell.execute_reply":"2021-06-30T18:58:23.051527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Saving","metadata":{}},{"cell_type":"code","source":"model.save('LSTM_Simple')\nmodel.save('LSTM_Simple.h5')","metadata":{"execution":{"iopub.status.busy":"2021-06-30T18:58:27.305782Z","iopub.execute_input":"2021-06-30T18:58:27.306722Z","iopub.status.idle":"2021-06-30T18:58:32.281374Z","shell.execute_reply.started":"2021-06-30T18:58:27.306644Z","shell.execute_reply":"2021-06-30T18:58:32.280298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"import os, fnmatch\ndef find(pattern, path):\n    result = []\n    for root, dirs, files in tqdm(os.walk(path),total=4369):\n        for name in files:\n            if fnmatch.fnmatch(name, pattern):\n                result.append(os.path.join(root, name))\n    return result\nTest_Dir = \"../input/g2net-gravitational-wave-detection/test\"\nnpy_files=find('*.npy', Test_Dir)\nprint(len(npy_files),\"Files Found!\")","metadata":{"execution":{"iopub.status.busy":"2021-06-30T19:10:29.854361Z","iopub.execute_input":"2021-06-30T19:10:29.854768Z","iopub.status.idle":"2021-06-30T19:10:34.252787Z","shell.execute_reply.started":"2021-06-30T19:10:29.854734Z","shell.execute_reply":"2021-06-30T19:10:34.25192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample pred\nx = np.ones((1,3,4096))\nsc = model.predict(x).flatten()[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T19:21:49.765116Z","iopub.execute_input":"2021-06-30T19:21:49.76549Z","iopub.status.idle":"2021-06-30T19:21:49.817506Z","shell.execute_reply.started":"2021-06-30T19:21:49.765459Z","shell.execute_reply":"2021-06-30T19:21:49.816202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_fileid(pth):\n    base = os.path.basename(pth)\n    return os.path.splitext(base)[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T19:31:35.845545Z","iopub.execute_input":"2021-06-30T19:31:35.84625Z","iopub.status.idle":"2021-06-30T19:31:35.853983Z","shell.execute_reply.started":"2021-06-30T19:31:35.846198Z","shell.execute_reply":"2021-06-30T19:31:35.852455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_id = []\nscores = []\nfor each_path in tqdm(npy_files):\n    x = np.load(each_path).reshape((1,3,4096))\n    sc = model.predict(x).flatten()[0]\n    img_id.append(get_fileid(each_path))\n    scores.append(scores)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T19:31:45.525021Z","iopub.execute_input":"2021-06-30T19:31:45.525601Z","iopub.status.idle":"2021-06-30T19:32:28.83158Z","shell.execute_reply.started":"2021-06-30T19:31:45.525566Z","shell.execute_reply":"2021-06-30T19:32:28.828682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.DataFrame.from_dict({\"id\": img_id,\n                                \"target\" : scores})\ndf_sub.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"Submission_Simple_LSTM.csv\",index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Evaluation (ToDo)\n\nRoC Curve, Area under RoC Curve","metadata":{}},{"cell_type":"code","source":"\"\"\"\n# https://www.kaggle.com/ihelon/gravitational-wave-exploratory-data-analysis\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nlist_y_true = [\n    [1., 1., 1., 1., 1., 1., 0., 0., 0., 0., 0., 0.],\n    [1., 1., 1., 1., 1., 1., 0., 0., 0., 0., 0., 0.],\n    [1., 1., 1., 1., 1., 1., 1., 1., 1., 1., 1., 0.], #  IMBALANCE\n]\nlist_y_pred = [\n    [0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 0.5],\n    [0.9, 0.9, 0.9, 0.9, 0.1, 0.9, 0.9, 0.1, 0.9, 0.1, 0.1, 0.5],\n    [1., 1., 1., 1., 1., 1., 1., 1., 1., 1., 1., 1.], #  IMBALANCE\n]\n\nfor y_true, y_pred in zip(list_y_true, list_y_pred):\n    fpr, tpr, _ = roc_curve(y_true, y_pred)\n    roc_auc = auc(fpr, tpr)\n\n    plt.figure(figsize=(5, 5))\n    plt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\n    plt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n    plt.xlim([-0.01, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver operating characteristic example')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n\"\"\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering (Todo)\n\nInitial Ideas\n\n- Peak Maxima\n- Peak Minima\n- Number of Maximas\n- Number of Minimas","metadata":{}},{"cell_type":"markdown","source":"# Zip the Output Directories","metadata":{}},{"cell_type":"code","source":"import os\nimport zipfile\nimport shutil\n\n#taken from : https://www.kaggle.com/xhlulu/recursion-2019-load-resize-and-save-images\n\ndef zip_and_remove(path):\n    ziph = zipfile.ZipFile(f'{path}.zip', 'w', zipfile.ZIP_DEFLATED)\n    \n    for root, dirs, files in os.walk(path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            ziph.write(file_path)\n            os.remove(file_path)\n    \n    ziph.close()\n    shutil.rmtree(path)\n    \n\nfor each_path in os.listdir(\"./\"):\n    if os.path.isdir(each_path):\n        zip_and_remove(each_path)","metadata":{},"execution_count":null,"outputs":[]}]}