{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport time\nimport dask\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom pathlib import Path\nfrom functools import partial\nimport matplotlib.pyplot as plt\n\n\nimport librosa\nfrom librosa import display\n\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\nsns.set()\n%config IPCompleter.use_jedi=False","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-03T17:46:52.881284Z","iopub.execute_input":"2021-07-03T17:46:52.881732Z","iopub.status.idle":"2021-07-03T17:46:55.674230Z","shell.execute_reply.started":"2021-07-03T17:46:52.881628Z","shell.execute_reply":"2021-07-03T17:46:55.673110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get the data","metadata":{}},{"cell_type":"code","source":"data_path = Path(\"../input/g2net-gravitational-wave-detection/\")\ntrain_npy_files_path = data_path / \"train\"\ntest_npy_files_path = data_path / \"test\"","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:46:58.921467Z","iopub.execute_input":"2021-07-03T17:46:58.921806Z","iopub.status.idle":"2021-07-03T17:46:58.926276Z","shell.execute_reply.started":"2021-07-03T17:46:58.921773Z","shell.execute_reply":"2021-07-03T17:46:58.925362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(data_path / \"training_labels.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:46:59.400933Z","iopub.execute_input":"2021-07-03T17:46:59.401315Z","iopub.status.idle":"2021-07-03T17:46:59.844284Z","shell.execute_reply.started":"2021-07-03T17:46:59.401285Z","shell.execute_reply":"2021-07-03T17:46:59.843298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(data_path / \"sample_submission.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:46:59.965047Z","iopub.execute_input":"2021-07-03T17:46:59.965369Z","iopub.status.idle":"2021-07-03T17:47:00.164140Z","shell.execute_reply.started":"2021-07-03T17:46:59.965342Z","shell.execute_reply":"2021-07-03T17:47:00.163321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_signal_path(signal_id, split=\"train\"):\n    if split == \"train\":\n        return str(train_npy_files_path / signal_id[0] / signal_id[1] / signal_id[2] / f\"{signal_id}.npy\")\n    elif split== \"test\":\n        return str(test_npy_files_path / signal_id[0] / signal_id[1] / signal_id[2] / f\"{signal_id}.npy\")","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:00.473289Z","iopub.execute_input":"2021-07-03T17:47:00.473640Z","iopub.status.idle":"2021-07-03T17:47:00.479311Z","shell.execute_reply.started":"2021-07-03T17:47:00.473605Z","shell.execute_reply":"2021-07-03T17:47:00.478586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\ntrain_df[\"filepath\"] = train_df[\"id\"].apply(partial(get_signal_path, split=\"train\"))\ntest_df[\"filepath\"] = test_df[\"id\"].apply(partial(get_signal_path, split=\"test\"))\nprint(f\"Filepaths stored in dataframes. Time taken: {time.time()-start:.2f} seconds\")","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:01.036144Z","iopub.execute_input":"2021-07-03T17:47:01.036678Z","iopub.status.idle":"2021-07-03T17:47:11.972240Z","shell.execute_reply.started":"2021-07-03T17:47:01.036633Z","shell.execute_reply":"2021-07-03T17:47:11.971317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Ananlysis","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:11.973937Z","iopub.execute_input":"2021-07-03T17:47:11.974305Z","iopub.status.idle":"2021-07-03T17:47:11.984295Z","shell.execute_reply.started":"2021-07-03T17:47:11.974265Z","shell.execute_reply":"2021-07-03T17:47:11.983350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Any duplicate signal in the data?\ntrain_df[\"id\"].duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:11.986052Z","iopub.execute_input":"2021-07-03T17:47:11.986412Z","iopub.status.idle":"2021-07-03T17:47:12.115448Z","shell.execute_reply.started":"2021-07-03T17:47:11.986370Z","shell.execute_reply":"2021-07-03T17:47:12.114709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Distribution of the labels\nplt.figure(figsize=(8, 5))\nsns.countplot(x=train_df[\"target\"], data=train_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:41.869843Z","iopub.execute_input":"2021-07-03T17:47:41.870305Z","iopub.status.idle":"2021-07-03T17:47:42.178928Z","shell.execute_reply.started":"2021-07-03T17:47:41.870275Z","shell.execute_reply":"2021-07-03T17:47:42.177844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The labels are almost equally distributed in the training set, a good sign. Let's check the actual counts of\nthe target labels to get a much better insight","metadata":{}},{"cell_type":"code","source":"train_df[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:47:44.883964Z","iopub.execute_input":"2021-07-03T17:47:44.884297Z","iopub.status.idle":"2021-07-03T17:47:44.898531Z","shell.execute_reply.started":"2021-07-03T17:47:44.884268Z","shell.execute_reply":"2021-07-03T17:47:44.897490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_traces(x, y, name, marker=None, color=None):\n    return go.Scatter(x=x,\n                      y=y,\n                      marker=marker,\n                      name=name,\n                     )","metadata":{"execution":{"iopub.status.busy":"2021-07-03T17:53:28.983846Z","iopub.execute_input":"2021-07-03T17:53:28.984254Z","iopub.status.idle":"2021-07-03T17:53:28.988121Z","shell.execute_reply.started":"2021-07-03T17:53:28.984221Z","shell.execute_reply":"2021-07-03T17:53:28.987446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_signals_from_array(signal_array,\n                            target,\n                            names=[\"LIGO Hanford\", \"LIGO Livingston\", \"Virgo\"],\n                            colors=[\"red\", \"green\", \"blue\"],\n                            markers=[None, None, None],\n                            subplots=False,\n                            title=None\n                           ):\n    \n    num_signals = len(signal_array)\n    \n    if num_signals > 1:\n        x = np.arange(len(signal_array[0]))\n        if not isinstance(target, list):\n            target = [target] * num_signals\n    else:\n        x = np.arange(len(signal_array))\n        \n    if num_signals > 1 and subplots:\n        fig = make_subplots(rows=num_signals, cols=1)\n        \n        for i in range(num_signals):\n            fig.add_trace(get_traces(x=x,\n                                     y=signal_array[i],\n                                     name=f\"target_{target[i]}:  {names[i]}\",\n                                     marker=markers[i],\n                                    ),\n                          row=i+1, col=1\n                         )\n    else:\n        fig = go.Figure()\n        \n        if num_signals > 1:\n            for i in range(num_signals):\n                fig.add_trace(get_traces(x=x,\n                                         y=signal_array[i],\n                                         name=f\"target_{target[i]}:  {names[i]}\",\n                                         marker=markers[i],\n                                        )\n                             )\n        else:\n            fig.add_trace(get_traces(x=x,\n                                     y=signal_array,\n                                     name=f\"target_{target}:  {names[0]}\",\n                                     marker=markers[0],\n                                    )\n                             )\n    if title:        \n        fig.layout.update(title_text=title, title_x=0.5)\n    return fig\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:16:56.709687Z","iopub.execute_input":"2021-07-03T18:16:56.710218Z","iopub.status.idle":"2021-07-03T18:16:56.718939Z","shell.execute_reply.started":"2021-07-03T18:16:56.710168Z","shell.execute_reply":"2021-07-03T18:16:56.718059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_values_from_df(df, idx):\n    _id = df[\"id\"][idx]\n    signal = np.load(df[\"filepath\"][idx])\n    target = df[\"target\"][idx]\n    return _id, signal, target","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:16:57.526372Z","iopub.execute_input":"2021-07-03T18:16:57.526733Z","iopub.status.idle":"2021-07-03T18:16:57.531296Z","shell.execute_reply.started":"2021-07-03T18:16:57.526699Z","shell.execute_reply":"2021-07-03T18:16:57.530367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_idx = np.random.randint(len(train_df))\nsample_id, sample_signal, sample_target = extract_values_from_df(train_df, random_idx)\n\nprint(\"Randomly chosen ID: \", sample_id)\nprint(\"Shape of the signal: \", sample_signal.shape)\nprint(\"Target for this signal: \", sample_target)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:16:58.342917Z","iopub.execute_input":"2021-07-03T18:16:58.343386Z","iopub.status.idle":"2021-07-03T18:16:58.364195Z","shell.execute_reply.started":"2021-07-03T18:16:58.343353Z","shell.execute_reply":"2021-07-03T18:16:58.363068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we can plot individual signals as well as combined. Let's do a comparison of the signal for different labels ","metadata":{}},{"cell_type":"code","source":"# Plot individual signals\nfig = plot_signals_from_array(\n    signal_array=sample_signal,\n    target=sample_target,\n    subplots=True,\n    title=f\"ID: {sample_id}  target: {sample_target}\"\n)\nfig.show()\n\n# Plot all signals combined\nfig = plot_signals_from_array(\n    signal_array=sample_signal,\n    target=sample_target,\n    title=f\"ID: {sample_id}  target: {sample_target}\"\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:17:15.603266Z","iopub.execute_input":"2021-07-03T18:17:15.603597Z","iopub.status.idle":"2021-07-03T18:17:15.692785Z","shell.execute_reply.started":"2021-07-03T18:17:15.603543Z","shell.execute_reply":"2021-07-03T18:17:15.691738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def random_sample_each_target(df):\n    # Pick a random index from all the indices where target == 0\n    random_idx_0 = np.random.choice(np.where(df[\"target\"]==0)[0])\n\n    # Extract values for this index from the dataframe\n    sample_id_0, sample_signal_0, sample_target_0 = extract_values_from_df(df, random_idx_0)\n\n    # Pick a random index form all the indices where target == 1\n    random_idx_1 = np.random.choice(np.where(df[\"target\"]==1)[0])\n\n    # Extract values for this index from the dataframe\n    sample_id_1, sample_signal_1, sample_target_1 = extract_values_from_df(df, random_idx_1)\n    \n    return (\n        [sample_id_0, sample_signal_0, sample_target_0],\n        [sample_id_1, sample_signal_1, sample_target_1]\n    )","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:38.780538Z","iopub.execute_input":"2021-07-03T18:04:38.780923Z","iopub.status.idle":"2021-07-03T18:04:38.786825Z","shell.execute_reply.started":"2021-07-03T18:04:38.780890Z","shell.execute_reply":"2021-07-03T18:04:38.785695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomly choosing samples for each target value\ntarget_0_sample, target_1_sample = random_sample_each_target(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:51.582732Z","iopub.execute_input":"2021-07-03T18:04:51.583040Z","iopub.status.idle":"2021-07-03T18:04:51.640349Z","shell.execute_reply.started":"2021-07-03T18:04:51.583012Z","shell.execute_reply":"2021-07-03T18:04:51.639334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare individual signals for different targets\nsignal_names = [\"LIGO Hanford\", \"LIGO Livingston\", \"Virgo\"]\nmarkers = [\n    [dict(color='rgba(255, 0, 0, 1.0)', size=10), dict(color='rgba(128, 50, 0, 0.8)', size=10)],\n    [dict(color='rgb(60, 179, 113, 1.0)', size=10), dict(color='rgba(25, 229, 206, 0.7)', size=10)],\n    [dict(color='rgba(0, 0, 255, 1.0)', size=10), dict(color='rgba(127, 0, 255, 0.5)', size=10)]\n]\n\nfor i in range(len(signal_names)):\n    fig = plot_signals_from_array(\n        signal_array=[target_0_sample[1][i], target_1_sample[1][i]],\n        target=[target_0_sample[2], target_1_sample[2]],\n        names=[signal_names[i], signal_names[i]],\n        markers=markers[i],\n        title=f\"target_0_ID: {target_0_sample[0]} \" \n              f\"target_1_ID: {target_1_sample[0]}\"\n    )\n    fig.show()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:53:03.009402Z","iopub.execute_input":"2021-07-03T18:53:03.009858Z","iopub.status.idle":"2021-07-03T18:53:03.079601Z","shell.execute_reply.started":"2021-07-03T18:53:03.009826Z","shell.execute_reply":"2021-07-03T18:53:03.078744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}