{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom random import shuffle\nimport matplotlib.pyplot as plt\nimport numba\nimport seaborn as sns\n\n# References: \n# https://www.kaggle.com/jsaguiar/seismic-data-exploration \n# https://www.kaggle.com/eylulyalcinkaya/exploratory-data-analysis\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nprint(os.listdir(\"../input\"))","execution_count":2,"outputs":[{"output_type":"stream","text":"['test', 'sample_submission.csv', 'train.csv']\n","name":"stdout"}]},{"metadata":{},"cell_type":"markdown","source":"<h2>2. Test data</h2>\n\nEach file from test folder corresponds to one prediction, i.e., seg_id from sample_submission."},{"metadata":{"trusted":true},"cell_type":"code","source":"test_folder_files = os.listdir(\"../input/test\")\n\nprint(\"\\nNumber of files in the test folder\", len(test_folder_files))","execution_count":3,"outputs":[{"output_type":"stream","text":"\nNumber of files in the test folder 2624\n","name":"stdout"}]},{"metadata":{},"cell_type":"markdown","source":"<h2>2. Training data</h2>\n\nAll the training data available in a csv file, with two columns:\n* Acoustic data (int16): the seismic signal\n* Time to failure (float64): the time until the next laboratory earthquake  (in seconds)"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\nprint(\"train shape\", train.shape)\npd.set_option(\"display.precision\", 15)  # show more decimals\ntrain.head()","execution_count":4,"outputs":[{"output_type":"stream","text":"train shape (629145480, 2)\n","name":"stdout"},{"output_type":"execute_result","execution_count":4,"data":{"text/plain":"   acoustic_data  time_to_failure\n0             12     1.4690999832\n1              6     1.4690999821\n2              8     1.4690999810\n3              5     1.4690999799\n4              8     1.4690999788","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>acoustic_data</th>\n      <th>time_to_failure</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>12</td>\n      <td>1.4690999832</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>6</td>\n      <td>1.4690999821</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>8</td>\n      <td>1.4690999810</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>5</td>\n      <td>1.4690999799</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>8</td>\n      <td>1.4690999788</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.acoustic_data.describe()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The time to failure (target value), in seconds, has max 16 seconds and minimum 1e-5 (approx. 0) "},{"metadata":{"trusted":true},"cell_type":"code","source":"@numba.jit\ndef get_stats(arr):\n    \"\"\"Memory efficient stats (min, max and mean). \"\"\"\n    size  = len(arr)\n    min_value = max_value = arr[0]\n    mean_value = 0\n    for i in numba.prange(size):\n        if arr[i] < min_value:\n            min_value = arr[i]\n        if arr[i] > max_value:\n            max_value = arr[i]\n        mean_value += arr[i]\n    return min_value, max_value, mean_value/size","execution_count":5,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmin, tmax, tmean = get_stats(train.acoustic_data.values)\nprint(\"min value: {:.6f}, max value: {:.2f}, mean: {:.4f}\".format(tmin, tmax, tmean))","execution_count":6,"outputs":[{"output_type":"stream","text":"min value: -5515.000000, max value: 5444.00, mean: 4.5195\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmin, tmax, tmean = get_stats(train.time_to_failure.values)\nprint(\"min value: {:.6f}, max value: {:.2f}, mean: {:.4f}\".format(tmin, tmax, tmean))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Plot all the data (629.145 million rows): (https://www.kaggle.com/jsaguiar/seismic-data-exploration)"},{"metadata":{"trusted":true},"cell_type":"code","source":"def single_timeseries(final_idx, init_idx=0, step=1, title=\"\",\n                      color1='orange', color2='g'):\n    idx = [i for i in range(init_idx, final_idx, step)]\n    fig, ax1 = plt.subplots(figsize=(10, 5))\n    fig.suptitle(title, fontsize=14)\n    \n    ax2 = ax1.twinx()\n    ax1.set_xlabel('index')\n    ax1.set_ylabel('Acoustic data')\n    ax2.set_ylabel('Time to failure')\n    p1 = sns.lineplot(data=train.iloc[idx].acoustic_data.values, ax=ax1, color=color1)\n    p2 = sns.lineplot(data=train.iloc[idx].time_to_failure.values, ax=ax2, color=color2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Observe that signal peaks usually happen before an earthquake.\nThere are 16 earthquakes in the training data. The shortest time to failure is 1.5 seconds for the first earthquake, while the longest is around 16 seconds."},{"metadata":{"trusted":true},"cell_type":"code","source":"single_timeseries(629145000, step=1000, title=\"Signal and time to failure with all training data\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}