{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![header](http://storage.googleapis.com/kaggle-competitions/kaggle/23249/logos/header.png)","metadata":{}},{"cell_type":"markdown","source":"## Competition\n\nIn this competition you are provided with a training set of time series data containing simulated gravitational wave measurements from a network of 3 gravitational wave interferometers (LIGO Hanford, LIGO Livingston, and Virgo). Each time series contains either detector noise or detector noise plus a simulated gravitational wave signal. The task is to identify when a signal is present in the data (target=1).\n\nEach data sample (npy file) contains 3 time series (1 for each detector) and each spans 2 sec and is sampled at 2,048 Hz.\n\n- train/ - the training set files, one npy file per observation; labels are provided in a files shown below\n- test/ - the test set files; you must predict the probability that the observation contains a gravitational wave\n- training_labels.csv - target values of whether the associated signal contains a gravitational wave\n- sample_submission.csv - a sample submission file in the correct format\n\nThe train/test paths are in the format:\n- '../input/g2net-gravitational-wave-detection/{train/test}/{id[0]}/{id[1]}/{id[2]}/{id}.npy'","metadata":{}},{"cell_type":"markdown","source":"## Evaluation Metrics\n\n- Area under the ROC curve","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\npd.set_option('display.max_columns', None)\npd.set_option('display.max_colwidth', None)\n#pd.set_option('display.max_rows', None)\n\nimport gc\n\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import classification_report, roc_auc_score, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom glob import glob\n\nfrom IPython.display import display\n\nplt.rcParams[\"figure.figsize\"] = (12, 8)\nplt.rcParams['axes.titlesize'] = 16\nplt.style.use('seaborn-whitegrid')\nsns.set_palette('Set2')\n\nimport os\nprint(os.listdir('/kaggle/input/g2net-gravitational-wave-detection/'))\n\nfrom time import time, strftime, gmtime\nstart = time()\nimport datetime\nprint(str(datetime.datetime.now()))\n\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-01T05:50:51.390523Z","iopub.execute_input":"2021-07-01T05:50:51.391020Z","iopub.status.idle":"2021-07-01T05:50:51.408147Z","shell.execute_reply.started":"2021-07-01T05:50:51.390961Z","shell.execute_reply":"2021-07-01T05:50:51.407105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/input/g2net-gravitational-wave-detection/'","metadata":{"execution":{"iopub.status.busy":"2021-07-01T05:50:51.712211Z","iopub.execute_input":"2021-07-01T05:50:51.712568Z","iopub.status.idle":"2021-07-01T05:50:51.717337Z","shell.execute_reply.started":"2021-07-01T05:50:51.712537Z","shell.execute_reply":"2021-07-01T05:50:51.716039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(base_dir + 'training_labels.csv')\nprint(train_labels.shape)\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T05:50:52.055350Z","iopub.execute_input":"2021-07-01T05:50:52.055875Z","iopub.status.idle":"2021-07-01T05:50:52.400248Z","shell.execute_reply.started":"2021-07-01T05:50:52.055841Z","shell.execute_reply":"2021-07-01T05:50:52.399498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(base_dir + 'sample_submission.csv')\nprint(sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T05:50:52.401443Z","iopub.execute_input":"2021-07-01T05:50:52.401854Z","iopub.status.idle":"2021-07-01T05:50:52.580222Z","shell.execute_reply.started":"2021-07-01T05:50:52.401825Z","shell.execute_reply":"2021-07-01T05:50:52.579438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Num of train files: {len(glob(base_dir + 'train/*/*/*/*.npy'))}\")\nprint(f\"Num of test files: {len(glob(base_dir + 'test/*/*/*/*.npy'))}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-01T05:50:52.724021Z","iopub.execute_input":"2021-07-01T05:50:52.724626Z","iopub.status.idle":"2021-07-01T05:51:42.007499Z","shell.execute_reply.started":"2021-07-01T05:50:52.724587Z","shell.execute_reply":"2021-07-01T05:51:42.006214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = sns.countplot(train_labels['target'])\nfor p in ax.patches:\n    ax.annotate(str(p.get_height()), (p.get_x() * 1.005, p.get_height() * 1.005))","metadata":{"execution":{"iopub.status.busy":"2021-07-01T05:51:42.009190Z","iopub.execute_input":"2021-07-01T05:51:42.009464Z","iopub.status.idle":"2021-07-01T05:51:42.172219Z","shell.execute_reply.started":"2021-07-01T05:51:42.009437Z","shell.execute_reply":"2021-07-01T05:51:42.171077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The target variables are well balanced!","metadata":{}},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"def image_id_to_path(img_id, flag = None):\n    path = f\"../input/g2net-gravitational-wave-detection/{flag}/{img_id[0]}/{img_id[1]}/{img_id[2]}/{img_id}.npy\"\n    return path\n\ndef visualize_waves(img_id, target, flag = None, plot = None):\n    plot_colors = [\"blue\", \"teal\", \"orange\"]\n    signal_path = image_id_to_path(img_id, flag)\n    signal = np.load(signal_path)\n    fig1, ax1 = plt.subplots(3, 1, figsize = (16, 8), sharex = True)\n    ax1 = ax1.ravel()\n    for i in range(len(ax1)):\n        ax1[i].plot(signal[i], color = plot_colors[i])\n        ax1[i].set_xlabel('Time (1/2048 sec)')\n    plt.suptitle(f\"Signal - Image_id: {img_id}, Target: {target}\")\n    plt.show()\n    \n    fig2, ax2 = plt.subplots(1, 3, figsize = (16, 6))\n    ax2 = ax2.ravel()\n    for j, ax in enumerate(ax2):\n        sns.kdeplot(signal[j], shade = True, color = plot_colors[j], ax = ax)\n    plt.suptitle(f\"Signal Distribution - Image_id: {img_id}, Target: {target}\")\n    plt.show()\n    \n    #Freq Spectrum of GW\n    fs = 2048\n    nfft = fs // 8\n    novl = nfft * 15 // 16\n    window = np.blackman(nfft)\n    spec_cmap = [\"inferno\", \"seismic\", \"icefire\"]\n    plt.figure(figsize = (16, 8))\n    for i in range(3):\n        plt.subplot(1, 3, i + 1)\n        plt.specgram(signal[i], NFFT = nfft, Fs = fs, window = window, \n                    noverlap = novl, cmap = spec_cmap[i], xextent = [0, 4000], vmin = -550, \n                     vmax = -440)\n        plt.grid(False)\n    plt.suptitle(f\"Signal Spectrogram- Image_id: {img_id}, Target: {target}\")\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-01T06:02:32.245666Z","iopub.execute_input":"2021-07-01T06:02:32.246023Z","iopub.status.idle":"2021-07-01T06:02:32.257585Z","shell.execute_reply.started":"2021-07-01T06:02:32.245992Z","shell.execute_reply":"2021-07-01T06:02:32.256146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check single npy file","metadata":{}},{"cell_type":"code","source":"sample_path = image_id_to_path(train_labels['id'][0], 'train')\nsample = np.load(sample_path)\nprint(sample.shape)\nplt.figure(figsize = (16, 8))\nplt.title('Sample Signal')\nplt.plot(sample)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T06:07:06.764704Z","iopub.execute_input":"2021-07-01T06:07:06.765058Z","iopub.status.idle":"2021-07-01T06:07:11.814569Z","shell.execute_reply.started":"2021-07-01T06:07:06.765029Z","shell.execute_reply":"2021-07-01T06:07:11.813655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize few samples","metadata":{}},{"cell_type":"code","source":"for idx in np.random.choice(train_labels.index, 4):\n    img_id = train_labels['id'].iloc[idx]\n    target = train_labels['target'].iloc[idx]\n    visualize_waves(img_id, target, 'train')","metadata":{"execution":{"iopub.status.busy":"2021-07-01T06:01:55.662965Z","iopub.execute_input":"2021-07-01T06:01:55.663299Z","iopub.status.idle":"2021-07-01T06:02:00.788178Z","shell.execute_reply.started":"2021-07-01T06:01:55.663271Z","shell.execute_reply":"2021-07-01T06:02:00.787116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# WIP","metadata":{}}]}