{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intro","metadata":{}},{"cell_type":"markdown","source":"There is a complex system of subfolders, while in the `training_labels.csv` each data point is referenced by an ID which corresponds to the file name. \n\nIn this notebook I map each ID to the full file path. I store both the corresponding dictionary mapping the ID to the full file path, as well as the enriched file `training_labels_with_paths.csv`.","metadata":{}},{"cell_type":"markdown","source":"# Mapping","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pickle\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nid_2_path = dict()\nfor dirname, _, filenames in tqdm(os.walk('/kaggle/input/g2net-gravitational-wave-detection/train/'), total=4369, desc='Checking filepath for each ID...'):\n    for filename in filenames:\n        if not os.path.isdir(filename):\n            id_2_path[os.path.splitext(filename)[0]] = os.path.join(dirname, filename)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-30T17:04:29.152741Z","iopub.execute_input":"2021-06-30T17:04:29.153108Z","iopub.status.idle":"2021-06-30T17:04:37.666845Z","shell.execute_reply.started":"2021-06-30T17:04:29.153078Z","shell.execute_reply":"2021-06-30T17:04:37.665817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Storing results","metadata":{}},{"cell_type":"code","source":"with open('id_2_path.pkl', 'wb') as f:\n    pickle.dump(id_2_path, f)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T16:58:29.994486Z","iopub.execute_input":"2021-06-30T16:58:29.994762Z","iopub.status.idle":"2021-06-30T16:58:30.546282Z","shell.execute_reply.started":"2021-06-30T16:58:29.994736Z","shell.execute_reply":"2021-06-30T16:58:30.545066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\ntraining_labels['filepath'] = training_labels['id'].map(id_2_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T16:58:30.547725Z","iopub.execute_input":"2021-06-30T16:58:30.548049Z","iopub.status.idle":"2021-06-30T16:58:31.472338Z","shell.execute_reply.started":"2021-06-30T16:58:30.548018Z","shell.execute_reply":"2021-06-30T16:58:31.471467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels.to_csv('training_labels_with_paths.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T16:58:31.473793Z","iopub.execute_input":"2021-06-30T16:58:31.474091Z","iopub.status.idle":"2021-06-30T16:58:34.465535Z","shell.execute_reply.started":"2021-06-30T16:58:31.474062Z","shell.execute_reply":"2021-06-30T16:58:34.464403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Demonstration","metadata":{}},{"cell_type":"markdown","source":"## Getting a subsample of positive data points","metadata":{}},{"cell_type":"markdown","source":"## Raw time series from LIGO Hanford","metadata":{}},{"cell_type":"code","source":"_, axs = plt.subplots(10, 2, figsize=(12, 30), sharex=True, sharey=True)\n\npos_subsample = training_labels.loc[training_labels['target'] == 1, 'filepath'].sample(10)\nneg_subsample = training_labels.loc[training_labels['target'] == 0, 'filepath'].sample(10)\n\nfor row_i, pos_filepath in enumerate(pos_subsample):\n    pos_data = np.load(pos_filepath)\n    axs[row_i, 0].plot(pos_data[0], c='r')\n    \nfor row_i, neg_filepath in enumerate(neg_subsample):\n    neg_data = np.load(neg_filepath)\n    axs[row_i, 1].plot(neg_data[0], c='b')\n\naxs[0, 0].set_title('Positives', fontsize=15)\naxs[0, 1].set_title('Negatives', fontsize=15)\nplt.suptitle('Visual comparison of randomly sampled positive and negative samples', fontsize=19)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T17:34:49.897183Z","iopub.execute_input":"2021-06-30T17:34:49.897588Z","iopub.status.idle":"2021-06-30T17:34:52.800171Z","shell.execute_reply.started":"2021-06-30T17:34:49.897554Z","shell.execute_reply":"2021-06-30T17:34:52.799422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}