{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-04-28T15:16:28.285804Z","iopub.execute_input":"2022-04-28T15:16:28.286098Z","iopub.status.idle":"2022-04-28T15:16:28.290897Z","shell.execute_reply.started":"2022-04-28T15:16:28.286064Z","shell.execute_reply":"2022-04-28T15:16:28.289855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import the Dataset","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a"}},{"cell_type":"code","source":"# Extract training data into a dataframe for further manipulation\ntrain = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', nrows=6000000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n\n# Print first 10 entries\ntrain.head(10)","metadata":{"_uuid":"d91c5674070f84a38a6e25b6ab8a58b34a27994e","execution":{"iopub.status.busy":"2022-04-28T15:04:15.445934Z","iopub.execute_input":"2022-04-28T15:04:15.446333Z","iopub.status.idle":"2022-04-28T15:04:17.535269Z","shell.execute_reply.started":"2022-04-28T15:04:15.446298Z","shell.execute_reply":"2022-04-28T15:04:17.534590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{"_uuid":"69f88065c08ee49b6addbef976eca917636dd576"}},{"cell_type":"code","source":"#visualize 1% of samples data, first 100 datapoints\ntrain_ad_sample_df = train['acoustic_data'].values[::100]\ntrain_ttf_sample_df = train['time_to_failure'].values[::100]\n\n#function for plotting based on both features\ndef plot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df, title=\"Acoustic data and time to failure: 1% sampled data\"):\n    fig, ax1 = plt.subplots(figsize=(12, 8))\n    plt.title(title)\n    plt.plot(train_ad_sample_df, color='r')\n    ax1.set_ylabel('acoustic data', color='r')\n    plt.legend(['acoustic data'], loc=(0.01, 0.95))\n    ax2 = ax1.twinx()\n    plt.plot(train_ttf_sample_df, color='b')\n    ax2.set_ylabel('time to failure', color='b')\n    plt.legend(['time to failure'], loc=(0.01, 0.9))\n    plt.grid(True)\n\nplot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df)\ndel train_ad_sample_df\ndel train_ttf_sample_df","metadata":{"_uuid":"ec82e1a615678bbde3f2f1d3524e3331fc79e398","execution":{"iopub.status.busy":"2022-04-28T15:04:24.861943Z","iopub.execute_input":"2022-04-28T15:04:24.862290Z","iopub.status.idle":"2022-04-28T15:04:25.371717Z","shell.execute_reply.started":"2022-04-28T15:04:24.862256Z","shell.execute_reply":"2022-04-28T15:04:25.371042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{"_uuid":"721073de2aa8c4ae5c557e0c6ee7c6b5f80eae7f"}},{"cell_type":"code","source":"# let's create a function to generate some statistical features based on the training data\ndef gen_features(X):\n    strain = []\n    strain.append(X.mean())\n    strain.append(X.std())\n    strain.append(X.min())\n    strain.append(X.max())\n    strain.append(X.kurtosis())\n    strain.append(X.skew())\n    strain.append(np.quantile(X,0.01))\n    strain.append(np.quantile(X,0.05))\n    strain.append(np.quantile(X,0.95))\n    strain.append(np.quantile(X,0.99))\n    strain.append(np.abs(X).max())\n    strain.append(np.abs(X).mean())\n    strain.append(np.abs(X).std())\n    return pd.Series(strain)","metadata":{"_uuid":"ec2bf170a4bc2af0a1c28f46c20b68ff5af11051","execution":{"iopub.status.busy":"2022-04-28T15:04:40.587427Z","iopub.execute_input":"2022-04-28T15:04:40.588040Z","iopub.status.idle":"2022-04-28T15:04:40.594819Z","shell.execute_reply.started":"2022-04-28T15:04:40.587989Z","shell.execute_reply":"2022-04-28T15:04:40.594147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For this, note how the chunks are used to iterate through this fairly large dataset. Each chunk is processed, and the results are appended to the *X_train* and *y_train* dataframes. This is somewhat different from some of the other kernels that I used, where all the chunks weren't iterated through and the training data was just limited.","metadata":{"_uuid":"414361a56616b9dce41424285ba0859ca5c8829f"}},{"cell_type":"code","source":"train = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', iterator=True, chunksize=150_000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n\nX_train = pd.DataFrame()\ny_train = pd.Series()\nfor df in train:\n    ch = gen_features(df['acoustic_data'])\n    X_train = X_train.append(ch, ignore_index=True)\n    y_train = y_train.append(pd.Series(df['time_to_failure'].values[-1]))\n    \n\n# Let's see what we have done\nX_train.describe()","metadata":{"_uuid":"f9b157d842d0c799a7b483f6bd327b18c1fe2e5a","execution":{"iopub.status.busy":"2022-04-28T15:08:05.739178Z","iopub.execute_input":"2022-04-28T15:08:05.739711Z","iopub.status.idle":"2022-04-28T15:11:09.283957Z","shell.execute_reply.started":"2022-04-28T15:08:05.739674Z","shell.execute_reply":"2022-04-28T15:11:09.283287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Implement Model","metadata":{"_uuid":"1b34ec7db9df118192efc89816e9fb3d7257b2ab"}},{"cell_type":"code","source":"# LSTM Model\nfrom sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import LSTM\n\nscaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\n\n# Reshape to correct dimensions\nX_train_scaled = X_train_scaled.reshape(X_train_scaled.shape[0],X_train_scaled.shape[1],1)\n\n# Model\nmodel = Sequential()\nmodel.add(LSTM(50,  input_shape=(X_train_scaled.shape[1], X_train_scaled.shape[2])))\nmodel.add(Dense(1))\nmodel.compile(loss='mae', optimizer='adam')\n","metadata":{"_uuid":"690627e7f3a11d5814b64bb062ebc8e85cab3aff","execution":{"iopub.status.busy":"2022-04-28T15:12:25.780147Z","iopub.execute_input":"2022-04-28T15:12:25.780432Z","iopub.status.idle":"2022-04-28T15:12:25.994749Z","shell.execute_reply.started":"2022-04-28T15:12:25.780399Z","shell.execute_reply":"2022-04-28T15:12:25.994028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit network\nhistory = model.fit(X_train_scaled, \n                    y_train, \n                    epochs=100,\n                    batch_size=64,\n                    verbose=0)\n\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T15:15:09.712899Z","iopub.execute_input":"2022-04-28T15:15:09.713611Z","iopub.status.idle":"2022-04-28T15:15:27.390481Z","shell.execute_reply.started":"2022-04-28T15:15:09.713573Z","shell.execute_reply":"2022-04-28T15:15:27.389700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T15:15:27.391941Z","iopub.execute_input":"2022-04-28T15:15:27.392234Z","iopub.status.idle":"2022-04-28T15:15:27.661547Z","shell.execute_reply.started":"2022-04-28T15:15:27.392197Z","shell.execute_reply":"2022-04-28T15:15:27.660881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate Model","metadata":{"_uuid":"94cb798c98bf452c5a81ee72ab0ab4960d05c5ce"}},{"cell_type":"code","source":"# Evaluate model\nfrom sklearn.metrics import mean_absolute_error\n    \ny_pred = model.predict(X_train_scaled)\nmae = mean_absolute_error(y_train, y_pred)\nprint('%.5f' % mae)","metadata":{"_uuid":"8a7b2318ebc1f38f7b1b478d7ad9b330591584c0","execution":{"iopub.status.busy":"2022-04-28T15:15:47.185761Z","iopub.execute_input":"2022-04-28T15:15:47.186369Z","iopub.status.idle":"2022-04-28T15:15:47.821628Z","shell.execute_reply.started":"2022-04-28T15:15:47.186329Z","shell.execute_reply":"2022-04-28T15:15:47.820876Z"},"trusted":true},"execution_count":null,"outputs":[]}]}