{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/LANL-Earthquake-Prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-03-10T16:16:16.171657Z","iopub.execute_input":"2023-03-10T16:16:16.172159Z","iopub.status.idle":"2023-03-10T16:16:58.257153Z","shell.execute_reply.started":"2023-03-10T16:16:16.172110Z","shell.execute_reply":"2023-03-10T16:16:58.256127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Introduction\n**^html link hidden here, fork notebook then click this cell to see it**\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../input/LANL-Earthquake-Prediction\"))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:16:58.258876Z","iopub.execute_input":"2023-03-10T16:16:58.259232Z","iopub.status.idle":"2023-03-10T16:16:58.266242Z","shell.execute_reply.started":"2023-03-10T16:16:58.259195Z","shell.execute_reply":"2023-03-10T16:16:58.265199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folder_files = os.listdir(\"../input/LANL-Earthquake-Prediction/test\")\nprint(test_folder_files[:10])  # print first 10\nprint(\"\\nNumber of files in the test folder\", len(test_folder_files))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:16:58.268125Z","iopub.execute_input":"2023-03-10T16:16:58.268564Z","iopub.status.idle":"2023-03-10T16:16:58.280133Z","shell.execute_reply.started":"2023-03-10T16:16:58.268471Z","shell.execute_reply":"2023-03-10T16:16:58.278910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/LANL-Earthquake-Prediction/sample_submission.csv')\nprint(\"Submission shape\", sample_sub.shape)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:16:58.282912Z","iopub.execute_input":"2023-03-10T16:16:58.283308Z","iopub.status.idle":"2023-03-10T16:16:58.327691Z","shell.execute_reply.started":"2023-03-10T16:16:58.283274Z","shell.execute_reply":"2023-03-10T16:16:58.326758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom tqdm import tqdm_notebook\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR, SVR\nfrom sklearn.metrics import mean_absolute_error\npd.options.display.precision = 15\n\nimport lightgbm as lgb\nimport xgboost as xgb\nimport time\nimport datetime\nfrom catboost import CatBoostRegressor\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold, KFold, RepeatedKFold\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.linear_model import LinearRegression\nimport gc\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom scipy.signal import hilbert\nfrom scipy.signal import hann\nfrom scipy.signal import convolve\nfrom scipy import stats\nfrom sklearn.kernel_ridge import KernelRidge","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:16:58.329177Z","iopub.execute_input":"2023-03-10T16:16:58.329553Z","iopub.status.idle":"2023-03-10T16:17:01.635842Z","shell.execute_reply.started":"2023-03-10T16:16:58.329512Z","shell.execute_reply":"2023-03-10T16:17:01.634784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float32})","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:17:01.637692Z","iopub.execute_input":"2023-03-10T16:17:01.638352Z","iopub.status.idle":"2023-03-10T16:20:59.865247Z","shell.execute_reply.started":"2023-03-10T16:17:01.638312Z","shell.execute_reply":"2023-03-10T16:20:59.863797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:20:59.867478Z","iopub.execute_input":"2023-03-10T16:20:59.868257Z","iopub.status.idle":"2023-03-10T16:20:59.883070Z","shell.execute_reply.started":"2023-03-10T16:20:59.868204Z","shell.execute_reply":"2023-03-10T16:20:59.881560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:20:59.884670Z","iopub.execute_input":"2023-03-10T16:20:59.885020Z","iopub.status.idle":"2023-03-10T16:20:59.892775Z","shell.execute_reply.started":"2023-03-10T16:20:59.884986Z","shell.execute_reply":"2023-03-10T16:20:59.891598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA : Distribution of Earthquakes","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"train_acoustic_data_small = train['acoustic_data'].values[::50]\ntrain_time_to_failure_small = train['time_to_failure'].values[::50]\n\nfig, ax1 = plt.subplots(figsize=(16, 8))\nplt.title(\"Trends of acoustic_data and time_to_failure. 2% of data (sampled)\")\nplt.plot(train_acoustic_data_small, color='b')\nax1.set_ylabel('acoustic_data', color='b')\nplt.legend(['acoustic_data'])\nax2 = ax1.twinx()\nplt.plot(train_time_to_failure_small, color='g')\nax2.set_ylabel('time_to_failure', color='g')\nplt.legend(['time_to_failure'], loc=(0.875, 0.9))\nplt.grid(False)\n\n\ndel train_acoustic_data_small\ndel train_time_to_failure_small","metadata":{"execution":{"iopub.status.busy":"2023-03-10T10:35:39.010988Z","iopub.execute_input":"2023-03-10T10:35:39.013098Z","iopub.status.idle":"2023-03-10T10:35:51.388135Z","shell.execute_reply.started":"2023-03-10T10:35:39.013049Z","shell.execute_reply":"2023-03-10T10:35:51.386109Z"}}},{"cell_type":"markdown","source":"train_acoustic_data_small = train['acoustic_data'].values[::100]\ntrain_time_to_failure_small = train['time_to_failure'].values[::100]\n\nfig, ax1 = plt.subplots(figsize=(16, 8))\nplt.title(\"Trends of acoustic_data and time_to_failure. 2% of data (sampled)\")\nplt.plot(train_acoustic_data_small, color='b')\nax1.set_ylabel('acoustic_data', color='b')\nplt.legend(['acoustic_data'])\nax2 = ax1.twinx()\nplt.plot(train_time_to_failure_small, color='g')\nax2.set_ylabel('time_to_failure', color='g')\nplt.legend(['time_to_failure'], loc=(0.875, 0.9))\nplt.grid(False)\n\ndel train_acoustic_data_small\ndel train_time_to_failure_small","metadata":{"execution":{"iopub.status.busy":"2023-03-10T10:35:51.393899Z","iopub.execute_input":"2023-03-10T10:35:51.394257Z","iopub.status.idle":"2023-03-10T10:35:57.964014Z","shell.execute_reply.started":"2023-03-10T10:35:51.394231Z","shell.execute_reply":"2023-03-10T10:35:57.962827Z"}}},{"cell_type":"markdown","source":"train_acoustic_data_small = train['acoustic_data'].values[::1000]\ntrain_time_to_failure_small = train['time_to_failure'].values[::1000]\n\nfig, ax1 = plt.subplots(figsize=(16, 8))\nplt.title(\"Trends of acoustic_data and time_to_failure. 2% of data (sampled)\")\nplt.plot(train_acoustic_data_small, color='b')\nax1.set_ylabel('acoustic_data', color='b')\nplt.legend(['acoustic_data'])\nax2 = ax1.twinx()\nplt.plot(train_time_to_failure_small, color='g')\nax2.set_ylabel('time_to_failure', color='g')\nplt.legend(['time_to_failure'], loc=(0.875, 0.9))\nplt.grid(False)\n\ndel train_acoustic_data_small\ndel train_time_to_failure_small","metadata":{"execution":{"iopub.status.busy":"2023-03-10T10:35:57.965224Z","iopub.execute_input":"2023-03-10T10:35:57.966887Z","iopub.status.idle":"2023-03-10T10:35:59.031953Z","shell.execute_reply.started":"2023-03-10T10:35:57.966821Z","shell.execute_reply":"2023-03-10T10:35:59.029810Z"}}},{"cell_type":"markdown","source":"train_acoustic_data_small = train['acoustic_data'].values[::150000]\ntrain_time_to_failure_small = train['time_to_failure'].values[::150000]\n\nfig, ax1 = plt.subplots(figsize=(16, 8))\nplt.title(\"Trends of acoustic_data and time_to_failure. 2% of data (sampled)\")\nplt.plot(train_acoustic_data_small, color='b')\nax1.set_ylabel('acoustic_data', color='b')\nplt.legend(['acoustic_data'])\nax2 = ax1.twinx()\nplt.plot(train_time_to_failure_small, color='g')\nax2.set_ylabel('time_to_failure', color='g')\nplt.legend(['time_to_failure'], loc=(0.875, 0.9))\nplt.grid(False)\n\ndel train_acoustic_data_small\ndel train_time_to_failure_small","metadata":{"execution":{"iopub.status.busy":"2023-03-10T10:35:59.034189Z","iopub.execute_input":"2023-03-10T10:35:59.035538Z","iopub.status.idle":"2023-03-10T10:35:59.440017Z","shell.execute_reply.started":"2023-03-10T10:35:59.035463Z","shell.execute_reply":"2023-03-10T10:35:59.438038Z"}}},{"cell_type":"markdown","source":"\n<a id=\"section-one\"></a>\n# Feature Creation\n**^html link hidden here, fork notebook then click this cell to see it**","metadata":{}},{"cell_type":"code","source":"# Create a training file with simple derived features\nrows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\ndef add_trend_feature(arr, abs_values=False):\n    idx = np.array(range(len(arr)))\n    if abs_values:\n        arr = np.abs(arr)\n    lr = LinearRegression()\n    lr.fit(idx.reshape(-1, 1), arr)\n    return lr.coef_[0]\n\ndef classic_sta_lta(x, length_sta, length_lta):\n    \n    sta = np.cumsum(x ** 2)\n\n    # Convert to float\n    sta = np.require(sta, dtype=np.float)\n\n    # Copy for LTA\n    lta = sta.copy()\n\n    # Compute the STA and the LTA\n    sta[length_sta:] = sta[length_sta:] - sta[:-length_sta]\n    sta /= length_sta\n    lta[length_lta:] = lta[length_lta:] - lta[:-length_lta]\n    lta /= length_lta\n\n    # Pad zeros\n    sta[:length_lta - 1] = 0\n\n    # Avoid division by zero by setting zero values to tiny float\n    dtiny = np.finfo(0.0).tiny\n    idx = lta < dtiny\n    lta[idx] = dtiny\n\n    return sta / lta\n\nX_tr = pd.DataFrame(index=range(segments), dtype=np.float64)\n\ny_tr = pd.DataFrame(index=range(segments), dtype=np.float64, columns=['time_to_failure'])\n\ntotal_mean = train['acoustic_data'].mean()\ntotal_std = train['acoustic_data'].std()\ntotal_max = train['acoustic_data'].max()\ntotal_min = train['acoustic_data'].min()\ntotal_sum = train['acoustic_data'].sum()\ntotal_abs_sum = np.abs(train['acoustic_data']).sum()\n\ndef calc_change_rate(x):\n    change = (np.diff(x) / x[:-1]).values\n    change = change[np.nonzero(change)[0]]\n    change = change[~np.isnan(change)]\n    change = change[change != -np.inf]\n    change = change[change != np.inf]\n    return np.mean(change)\n\nfor segment in tqdm_notebook(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = pd.Series(seg['acoustic_data'].values)\n    y = seg['time_to_failure'].values[-1]\n    \n    y_tr.loc[segment, 'time_to_failure'] = y\n    X_tr.loc[segment, 'mean'] = x.mean()\n    X_tr.loc[segment, 'std'] = x.std()\n    X_tr.loc[segment, 'max'] = x.max()\n    X_tr.loc[segment, 'min'] = x.min()\n    \n    X_tr.loc[segment, 'mean_change_abs'] = np.mean(np.diff(x))\n    X_tr.loc[segment, 'mean_change_rate'] = calc_change_rate(x)\n    X_tr.loc[segment, 'abs_max'] = np.abs(x).max()\n    X_tr.loc[segment, 'abs_min'] = np.abs(x).min()\n    \n    X_tr.loc[segment, 'std_first_50000'] = x[:50000].std()\n    X_tr.loc[segment, 'std_last_50000'] = x[-50000:].std()\n    X_tr.loc[segment, 'std_first_10000'] = x[:10000].std()\n    X_tr.loc[segment, 'std_last_10000'] = x[-10000:].std()\n    \n    X_tr.loc[segment, 'avg_first_50000'] = x[:50000].mean()\n    X_tr.loc[segment, 'avg_last_50000'] = x[-50000:].mean()\n    X_tr.loc[segment, 'avg_first_10000'] = x[:10000].mean()\n    X_tr.loc[segment, 'avg_last_10000'] = x[-10000:].mean()\n    \n    X_tr.loc[segment, 'min_first_50000'] = x[:50000].min()\n    X_tr.loc[segment, 'min_last_50000'] = x[-50000:].min()\n    X_tr.loc[segment, 'min_first_10000'] = x[:10000].min()\n    X_tr.loc[segment, 'min_last_10000'] = x[-10000:].min()\n    \n    X_tr.loc[segment, 'max_first_50000'] = x[:50000].max()\n    X_tr.loc[segment, 'max_last_50000'] = x[-50000:].max()\n    X_tr.loc[segment, 'max_first_10000'] = x[:10000].max()\n    X_tr.loc[segment, 'max_last_10000'] = x[-10000:].max()\n    \n    X_tr.loc[segment, 'max_to_min'] = x.max() / np.abs(x.min())\n    X_tr.loc[segment, 'max_to_min_diff'] = x.max() - np.abs(x.min())\n    X_tr.loc[segment, 'count_big'] = len(x[np.abs(x) > 500])\n    X_tr.loc[segment, 'sum'] = x.sum()\n    \n    X_tr.loc[segment, 'mean_change_rate_first_50000'] = calc_change_rate(x[:50000])\n    X_tr.loc[segment, 'mean_change_rate_last_50000'] = calc_change_rate(x[-50000:])\n    X_tr.loc[segment, 'mean_change_rate_first_10000'] = calc_change_rate(x[:10000])\n    X_tr.loc[segment, 'mean_change_rate_last_10000'] = calc_change_rate(x[-10000:])\n    \n    X_tr.loc[segment, 'q95'] = np.quantile(x, 0.95)\n    X_tr.loc[segment, 'q99'] = np.quantile(x, 0.99)\n    X_tr.loc[segment, 'q05'] = np.quantile(x, 0.05)\n    X_tr.loc[segment, 'q01'] = np.quantile(x, 0.01)\n    \n    X_tr.loc[segment, 'abs_q95'] = np.quantile(np.abs(x), 0.95)\n    X_tr.loc[segment, 'abs_q99'] = np.quantile(np.abs(x), 0.99)\n    X_tr.loc[segment, 'abs_q05'] = np.quantile(np.abs(x), 0.05)\n    X_tr.loc[segment, 'abs_q01'] = np.quantile(np.abs(x), 0.01)\n    \n    X_tr.loc[segment, 'trend'] = add_trend_feature(x)\n    X_tr.loc[segment, 'abs_trend'] = add_trend_feature(x, abs_values=True)\n    X_tr.loc[segment, 'abs_mean'] = np.abs(x).mean()\n    X_tr.loc[segment, 'abs_std'] = np.abs(x).std()\n    \n    X_tr.loc[segment, 'mad'] = x.mad()\n    X_tr.loc[segment, 'kurt'] = x.kurtosis()\n    X_tr.loc[segment, 'skew'] = x.skew()\n    X_tr.loc[segment, 'med'] = x.median()\n    \n    X_tr.loc[segment, 'Hilbert_mean'] = np.abs(hilbert(x)).mean()\n    X_tr.loc[segment, 'Hann_window_mean'] = (convolve(x, hann(150), mode='same') / sum(hann(150))).mean()\n    X_tr.loc[segment, 'classic_sta_lta1_mean'] = classic_sta_lta(x, 500, 10000).mean()\n    X_tr.loc[segment, 'classic_sta_lta2_mean'] = classic_sta_lta(x, 5000, 100000).mean()\n    X_tr.loc[segment, 'classic_sta_lta3_mean'] = classic_sta_lta(x, 3333, 6666).mean()\n    X_tr.loc[segment, 'classic_sta_lta4_mean'] = classic_sta_lta(x, 10000, 25000).mean()\n    X_tr.loc[segment, 'classic_sta_lta5_mean'] = classic_sta_lta(x, 50, 1000).mean()\n    X_tr.loc[segment, 'classic_sta_lta6_mean'] = classic_sta_lta(x, 100, 5000).mean()\n    X_tr.loc[segment, 'classic_sta_lta7_mean'] = classic_sta_lta(x, 333, 666).mean()\n    X_tr.loc[segment, 'classic_sta_lta8_mean'] = classic_sta_lta(x, 4000, 10000).mean()\n    X_tr.loc[segment, 'Moving_average_700_mean'] = x.rolling(window=700).mean().mean(skipna=True)\n    ewma = pd.Series.ewm\n    X_tr.loc[segment, 'exp_Moving_average_300_mean'] = (ewma(x, span=300).mean()).mean(skipna=True)\n    X_tr.loc[segment, 'exp_Moving_average_3000_mean'] = ewma(x, span=3000).mean().mean(skipna=True)\n    X_tr.loc[segment, 'exp_Moving_average_30000_mean'] = ewma(x, span=30000).mean().mean(skipna=True)\n    no_of_std = 3\n    X_tr.loc[segment, 'MA_700MA_std_mean'] = x.rolling(window=700).std().mean()\n    X_tr.loc[segment,'MA_700MA_BB_high_mean'] = (X_tr.loc[segment, 'Moving_average_700_mean'] + no_of_std * X_tr.loc[segment, 'MA_700MA_std_mean']).mean()\n    X_tr.loc[segment,'MA_700MA_BB_low_mean'] = (X_tr.loc[segment, 'Moving_average_700_mean'] - no_of_std * X_tr.loc[segment, 'MA_700MA_std_mean']).mean()\n    X_tr.loc[segment, 'MA_400MA_std_mean'] = x.rolling(window=400).std().mean()\n    X_tr.loc[segment,'MA_400MA_BB_high_mean'] = (X_tr.loc[segment, 'Moving_average_700_mean'] + no_of_std * X_tr.loc[segment, 'MA_400MA_std_mean']).mean()\n    X_tr.loc[segment,'MA_400MA_BB_low_mean'] = (X_tr.loc[segment, 'Moving_average_700_mean'] - no_of_std * X_tr.loc[segment, 'MA_400MA_std_mean']).mean()\n    X_tr.loc[segment, 'MA_1000MA_std_mean'] = x.rolling(window=1000).std().mean()\n    X_tr.drop('Moving_average_700_mean', axis=1, inplace=True)\n    \n    X_tr.loc[segment, 'iqr'] = np.subtract(*np.percentile(x, [75, 25]))\n    X_tr.loc[segment, 'q999'] = np.quantile(x,0.999)\n    X_tr.loc[segment, 'q001'] = np.quantile(x,0.001)\n    X_tr.loc[segment, 'ave10'] = stats.trim_mean(x, 0.1)\n\n    for windows in [10, 100, 1000]:\n        x_roll_std = x.rolling(windows).std().dropna().values\n        x_roll_mean = x.rolling(windows).mean().dropna().values\n        \n        X_tr.loc[segment, 'ave_roll_std_' + str(windows)] = x_roll_std.mean()\n        X_tr.loc[segment, 'std_roll_std_' + str(windows)] = x_roll_std.std()\n        X_tr.loc[segment, 'max_roll_std_' + str(windows)] = x_roll_std.max()\n        X_tr.loc[segment, 'min_roll_std_' + str(windows)] = x_roll_std.min()\n        X_tr.loc[segment, 'q01_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.01)\n        X_tr.loc[segment, 'q05_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.05)\n        X_tr.loc[segment, 'q95_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.95)\n        X_tr.loc[segment, 'q99_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.99)\n        X_tr.loc[segment, 'av_change_abs_roll_std_' + str(windows)] = np.mean(np.diff(x_roll_std))\n        X_tr.loc[segment, 'av_change_rate_roll_std_' + str(windows)] = np.mean(np.nonzero((np.diff(x_roll_std) / x_roll_std[:-1]))[0])\n        X_tr.loc[segment, 'abs_max_roll_std_' + str(windows)] = np.abs(x_roll_std).max()\n        \n        X_tr.loc[segment, 'ave_roll_mean_' + str(windows)] = x_roll_mean.mean()\n        X_tr.loc[segment, 'std_roll_mean_' + str(windows)] = x_roll_mean.std()\n        X_tr.loc[segment, 'max_roll_mean_' + str(windows)] = x_roll_mean.max()\n        X_tr.loc[segment, 'min_roll_mean_' + str(windows)] = x_roll_mean.min()\n        X_tr.loc[segment, 'q01_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.01)\n        X_tr.loc[segment, 'q05_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.05)\n        X_tr.loc[segment, 'q95_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.95)\n        X_tr.loc[segment, 'q99_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.99)\n        X_tr.loc[segment, 'av_change_abs_roll_mean_' + str(windows)] = np.mean(np.diff(x_roll_mean))\n        X_tr.loc[segment, 'av_change_rate_roll_mean_' + str(windows)] = np.mean(np.nonzero((np.diff(x_roll_mean) / x_roll_mean[:-1]))[0])\n        X_tr.loc[segment, 'abs_max_roll_mean_' + str(windows)] = np.abs(x_roll_mean).max()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:20:59.895522Z","iopub.execute_input":"2023-03-10T16:20:59.896404Z","iopub.status.idle":"2023-03-10T16:39:47.308091Z","shell.execute_reply.started":"2023-03-10T16:20:59.896361Z","shell.execute_reply":"2023-03-10T16:39:47.306193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:39:47.313617Z","iopub.execute_input":"2023-03-10T16:39:47.314044Z","iopub.status.idle":"2023-03-10T16:39:47.325624Z","shell.execute_reply.started":"2023-03-10T16:39:47.314005Z","shell.execute_reply":"2023-03-10T16:39:47.324037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:39:47.327047Z","iopub.execute_input":"2023-03-10T16:39:47.327397Z","iopub.status.idle":"2023-03-10T16:39:47.367094Z","shell.execute_reply.started":"2023-03-10T16:39:47.327363Z","shell.execute_reply":"2023-03-10T16:39:47.365670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr.info(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:44:37.498916Z","iopub.execute_input":"2023-03-10T16:44:37.499393Z","iopub.status.idle":"2023-03-10T16:44:37.515072Z","shell.execute_reply.started":"2023-03-10T16:44:37.499358Z","shell.execute_reply":"2023-03-10T16:44:37.513761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.abs(X_tr.corrwith(y_tr['time_to_failure'])).sort_values(ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:39:47.405076Z","iopub.execute_input":"2023-03-10T16:39:47.406108Z","iopub.status.idle":"2023-03-10T16:39:47.477455Z","shell.execute_reply.started":"2023-03-10T16:39:47.406037Z","shell.execute_reply":"2023-03-10T16:39:47.476114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"plt.figure(figsize=(44, 24))\ncols = list(np.abs(X_tr.corrwith(y_tr['time_to_failure'])).sort_values(ascending=False).head(24).index)\nfor i, col in enumerate(cols):\n    plt.subplot(6, 4, i + 1)\n    plt.plot(X_tr[col], color='blue')\n    plt.title(col)\n    ax1.set_ylabel(col, color='b')\n\n    ax2 = ax1.twinx()\n    plt.plot(y_tr, color='g')\n    ax2.set_ylabel('time_to_failure', color='g')\n    plt.legend([col, 'time_to_failure'], loc=(0.875, 0.9))\n    plt.grid(False)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:39:47.480028Z","iopub.execute_input":"2023-03-10T16:39:47.480627Z","iopub.status.idle":"2023-03-10T16:39:48.058917Z","shell.execute_reply.started":"2023-03-10T16:39:47.480569Z","shell.execute_reply":"2023-03-10T16:39:48.055846Z"}}},{"cell_type":"code","source":"means_dict = {}\nfor col in X_tr.columns:\n    if X_tr[col].isnull().any():\n        print(col)\n        mean_value = X_tr.loc[X_tr[col] != -np.inf, col].mean()\n        X_tr.loc[X_tr[col] == -np.inf, col] = mean_value\n        X_tr[col] = X_tr[col].fillna(mean_value)\n        means_dict[col] = mean_value","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:44:53.076299Z","iopub.execute_input":"2023-03-10T16:44:53.076737Z","iopub.status.idle":"2023-03-10T16:44:53.112677Z","shell.execute_reply.started":"2023-03-10T16:44:53.076700Z","shell.execute_reply":"2023-03-10T16:44:53.111468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_tr)\nX_train_scaled = pd.DataFrame(scaler.transform(X_tr), columns=X_tr.columns)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:44:55.346596Z","iopub.execute_input":"2023-03-10T16:44:55.347874Z","iopub.status.idle":"2023-03-10T16:44:55.370536Z","shell.execute_reply.started":"2023-03-10T16:44:55.347817Z","shell.execute_reply":"2023-03-10T16:44:55.369237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../input/LANL-Earthquake-Prediction/\"))\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:44:57.263906Z","iopub.execute_input":"2023-03-10T16:44:57.265181Z","iopub.status.idle":"2023-03-10T16:44:57.272066Z","shell.execute_reply.started":"2023-03-10T16:44:57.265121Z","shell.execute_reply":"2023-03-10T16:44:57.270625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/LANL-Earthquake-Prediction/sample_submission.csv', index_col='seg_id')\nX_test = pd.DataFrame(columns=X_tr.columns, dtype=np.float64, index=submission.index)\nplt.figure(figsize=(22, 16))\n\nfor i, seg_id in enumerate(tqdm_notebook(X_test.index)):\n    seg = pd.read_csv('../input/LANL-Earthquake-Prediction/test/' + seg_id + '.csv')\n    \n    x = pd.Series(seg['acoustic_data'].values)\n    X_test.loc[seg_id, 'mean'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()\n        \n    X_test.loc[seg_id, 'mean_change_abs'] = np.mean(np.diff(x))\n    X_test.loc[seg_id, 'mean_change_rate'] = calc_change_rate(x)\n    X_test.loc[seg_id, 'abs_max'] = np.abs(x).max()\n    X_test.loc[seg_id, 'abs_min'] = np.abs(x).min()\n    \n    X_test.loc[seg_id, 'std_first_50000'] = x[:50000].std()\n    X_test.loc[seg_id, 'std_last_50000'] = x[-50000:].std()\n    X_test.loc[seg_id, 'std_first_10000'] = x[:10000].std()\n    X_test.loc[seg_id, 'std_last_10000'] = x[-10000:].std()\n    \n    X_test.loc[seg_id, 'avg_first_50000'] = x[:50000].mean()\n    X_test.loc[seg_id, 'avg_last_50000'] = x[-50000:].mean()\n    X_test.loc[seg_id, 'avg_first_10000'] = x[:10000].mean()\n    X_test.loc[seg_id, 'avg_last_10000'] = x[-10000:].mean()\n    \n    X_test.loc[seg_id, 'min_first_50000'] = x[:50000].min()\n    X_test.loc[seg_id, 'min_last_50000'] = x[-50000:].min()\n    X_test.loc[seg_id, 'min_first_10000'] = x[:10000].min()\n    X_test.loc[seg_id, 'min_last_10000'] = x[-10000:].min()\n    \n    X_test.loc[seg_id, 'max_first_50000'] = x[:50000].max()\n    X_test.loc[seg_id, 'max_last_50000'] = x[-50000:].max()\n    X_test.loc[seg_id, 'max_first_10000'] = x[:10000].max()\n    X_test.loc[seg_id, 'max_last_10000'] = x[-10000:].max()\n    \n    X_test.loc[seg_id, 'max_to_min'] = x.max() / np.abs(x.min())\n    X_test.loc[seg_id, 'max_to_min_diff'] = x.max() - np.abs(x.min())\n    X_test.loc[seg_id, 'count_big'] = len(x[np.abs(x) > 500])\n    X_test.loc[seg_id, 'sum'] = x.sum()\n    \n    X_test.loc[seg_id, 'mean_change_rate_first_50000'] = calc_change_rate(x[:50000])\n    X_test.loc[seg_id, 'mean_change_rate_last_50000'] = calc_change_rate(x[-50000:])\n    X_test.loc[seg_id, 'mean_change_rate_first_10000'] = calc_change_rate(x[:10000])\n    X_test.loc[seg_id, 'mean_change_rate_last_10000'] = calc_change_rate(x[-10000:])\n    \n    X_test.loc[seg_id, 'q95'] = np.quantile(x,0.95)\n    X_test.loc[seg_id, 'q99'] = np.quantile(x,0.99)\n    X_test.loc[seg_id, 'q05'] = np.quantile(x,0.05)\n    X_test.loc[seg_id, 'q01'] = np.quantile(x,0.01)\n    \n    X_test.loc[seg_id, 'abs_q95'] = np.quantile(np.abs(x), 0.95)\n    X_test.loc[seg_id, 'abs_q99'] = np.quantile(np.abs(x), 0.99)\n    X_test.loc[seg_id, 'abs_q05'] = np.quantile(np.abs(x), 0.05)\n    X_test.loc[seg_id, 'abs_q01'] = np.quantile(np.abs(x), 0.01)\n    \n    X_test.loc[seg_id, 'trend'] = add_trend_feature(x)\n    X_test.loc[seg_id, 'abs_trend'] = add_trend_feature(x, abs_values=True)\n    X_test.loc[seg_id, 'abs_mean'] = np.abs(x).mean()\n    X_test.loc[seg_id, 'abs_std'] = np.abs(x).std()\n    \n    X_test.loc[seg_id, 'mad'] = x.mad()\n    X_test.loc[seg_id, 'kurt'] = x.kurtosis()\n    X_test.loc[seg_id, 'skew'] = x.skew()\n    X_test.loc[seg_id, 'med'] = x.median()\n    \n    X_test.loc[seg_id, 'Hilbert_mean'] = np.abs(hilbert(x)).mean()\n    X_test.loc[seg_id, 'Hann_window_mean'] = (convolve(x, hann(150), mode='same') / sum(hann(150))).mean()\n    X_test.loc[seg_id, 'classic_sta_lta1_mean'] = classic_sta_lta(x, 500, 10000).mean()\n    X_test.loc[seg_id, 'classic_sta_lta2_mean'] = classic_sta_lta(x, 5000, 100000).mean()\n    X_test.loc[seg_id, 'classic_sta_lta3_mean'] = classic_sta_lta(x, 3333, 6666).mean()\n    X_test.loc[seg_id, 'classic_sta_lta4_mean'] = classic_sta_lta(x, 10000, 25000).mean()\n    X_test.loc[seg_id, 'classic_sta_lta5_mean'] = classic_sta_lta(x, 50, 1000).mean()\n    X_test.loc[seg_id, 'classic_sta_lta6_mean'] = classic_sta_lta(x, 100, 5000).mean()\n    X_test.loc[seg_id, 'classic_sta_lta7_mean'] = classic_sta_lta(x, 333, 666).mean()\n    X_test.loc[seg_id, 'classic_sta_lta8_mean'] = classic_sta_lta(x, 4000, 10000).mean()\n    X_test.loc[seg_id, 'Moving_average_700_mean'] = x.rolling(window=700).mean().mean(skipna=True)\n    ewma = pd.Series.ewm\n    X_test.loc[seg_id, 'exp_Moving_average_300_mean'] = (ewma(x, span=300).mean()).mean(skipna=True)\n    X_test.loc[seg_id, 'exp_Moving_average_3000_mean'] = ewma(x, span=3000).mean().mean(skipna=True)\n    X_test.loc[seg_id, 'exp_Moving_average_30000_mean'] = ewma(x, span=30000).mean().mean(skipna=True)\n    no_of_std = 3\n    X_test.loc[seg_id, 'MA_700MA_std_mean'] = x.rolling(window=700).std().mean()\n    X_test.loc[seg_id,'MA_700MA_BB_high_mean'] = (X_test.loc[seg_id, 'Moving_average_700_mean'] + no_of_std * X_test.loc[seg_id, 'MA_700MA_std_mean']).mean()\n    X_test.loc[seg_id,'MA_700MA_BB_low_mean'] = (X_test.loc[seg_id, 'Moving_average_700_mean'] - no_of_std * X_test.loc[seg_id, 'MA_700MA_std_mean']).mean()\n    X_test.loc[seg_id, 'MA_400MA_std_mean'] = x.rolling(window=400).std().mean()\n    X_test.loc[seg_id,'MA_400MA_BB_high_mean'] = (X_test.loc[seg_id, 'Moving_average_700_mean'] + no_of_std * X_test.loc[seg_id, 'MA_400MA_std_mean']).mean()\n    X_test.loc[seg_id,'MA_400MA_BB_low_mean'] = (X_test.loc[seg_id, 'Moving_average_700_mean'] - no_of_std * X_test.loc[seg_id, 'MA_400MA_std_mean']).mean()\n    X_test.loc[seg_id, 'MA_1000MA_std_mean'] = x.rolling(window=1000).std().mean()\n    X_test.drop('Moving_average_700_mean', axis=1, inplace=True)\n    \n    X_test.loc[seg_id, 'iqr'] = np.subtract(*np.percentile(x, [75, 25]))\n    X_test.loc[seg_id, 'q999'] = np.quantile(x,0.999)\n    X_test.loc[seg_id, 'q001'] = np.quantile(x,0.001)\n    X_test.loc[seg_id, 'ave10'] = stats.trim_mean(x, 0.1)\n    \n    for windows in [10, 100, 1000]:\n        x_roll_std = x.rolling(windows).std().dropna().values\n        x_roll_mean = x.rolling(windows).mean().dropna().values\n        \n        X_test.loc[seg_id, 'ave_roll_std_' + str(windows)] = x_roll_std.mean()\n        X_test.loc[seg_id, 'std_roll_std_' + str(windows)] = x_roll_std.std()\n        X_test.loc[seg_id, 'max_roll_std_' + str(windows)] = x_roll_std.max()\n        X_test.loc[seg_id, 'min_roll_std_' + str(windows)] = x_roll_std.min()\n        X_test.loc[seg_id, 'q01_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.01)\n        X_test.loc[seg_id, 'q05_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.05)\n        X_test.loc[seg_id, 'q95_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.95)\n        X_test.loc[seg_id, 'q99_roll_std_' + str(windows)] = np.quantile(x_roll_std, 0.99)\n        X_test.loc[seg_id, 'av_change_abs_roll_std_' + str(windows)] = np.mean(np.diff(x_roll_std))\n        X_test.loc[seg_id, 'av_change_rate_roll_std_' + str(windows)] = np.mean(np.nonzero((np.diff(x_roll_std) / x_roll_std[:-1]))[0])\n        X_test.loc[seg_id, 'abs_max_roll_std_' + str(windows)] = np.abs(x_roll_std).max()\n        \n        X_test.loc[seg_id, 'ave_roll_mean_' + str(windows)] = x_roll_mean.mean()\n        X_test.loc[seg_id, 'std_roll_mean_' + str(windows)] = x_roll_mean.std()\n        X_test.loc[seg_id, 'max_roll_mean_' + str(windows)] = x_roll_mean.max()\n        X_test.loc[seg_id, 'min_roll_mean_' + str(windows)] = x_roll_mean.min()\n        X_test.loc[seg_id, 'q01_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.01)\n        X_test.loc[seg_id, 'q05_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.05)\n        X_test.loc[seg_id, 'q95_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.95)\n        X_test.loc[seg_id, 'q99_roll_mean_' + str(windows)] = np.quantile(x_roll_mean, 0.99)\n        X_test.loc[seg_id, 'av_change_abs_roll_mean_' + str(windows)] = np.mean(np.diff(x_roll_mean))\n        X_test.loc[seg_id, 'av_change_rate_roll_mean_' + str(windows)] = np.mean(np.nonzero((np.diff(x_roll_mean) / x_roll_mean[:-1]))[0])\n        X_test.loc[seg_id, 'abs_max_roll_mean_' + str(windows)] = np.abs(x_roll_mean).max()\n    \n    if i < 12:\n        plt.subplot(6, 4, i + 1)\n        plt.plot(seg['acoustic_data'])\n        plt.title(seg_id)\n\nfor col in X_test.columns:\n    if X_test[col].isnull().any():\n        X_test.loc[X_test[col] == -np.inf, col] = means_dict[col]\n        X_test[col] = X_test[col].fillna(means_dict[col])\n        \nX_test_scaled = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:45:00.503524Z","iopub.execute_input":"2023-03-10T16:45:00.503978Z","iopub.status.idle":"2023-03-10T16:58:47.312517Z","shell.execute_reply.started":"2023-03-10T16:45:00.503938Z","shell.execute_reply":"2023-03-10T16:58:47.310863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:47.314744Z","iopub.execute_input":"2023-03-10T16:58:47.315094Z","iopub.status.idle":"2023-03-10T16:58:47.322670Z","shell.execute_reply.started":"2023-03-10T16:58:47.315062Z","shell.execute_reply":"2023-03-10T16:58:47.321279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_tr.shape ","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:47.324591Z","iopub.execute_input":"2023-03-10T16:58:47.324966Z","iopub.status.idle":"2023-03-10T16:58:47.339141Z","shell.execute_reply.started":"2023-03-10T16:58:47.324930Z","shell.execute_reply":"2023-03-10T16:58:47.337170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled.shape,X_test_scaled.shape,y_tr.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:47.343446Z","iopub.execute_input":"2023-03-10T16:58:47.344061Z","iopub.status.idle":"2023-03-10T16:58:47.356758Z","shell.execute_reply.started":"2023-03-10T16:58:47.343999Z","shell.execute_reply":"2023-03-10T16:58:47.355109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Feature selection \n**^html link hidden here, fork notebook then click this cell to see it**","metadata":{}},{"cell_type":"code","source":"type(X_train_scaled),type(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:39:48.076499Z","iopub.status.idle":"2023-03-10T16:39:48.076899Z","shell.execute_reply.started":"2023-03-10T16:39:48.076693Z","shell.execute_reply":"2023-03-10T16:39:48.076713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Model\n**^html link hidden here, fork notebook then click this cell to see it**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import GRU\n\nfrom tensorflow.keras.layers import Dense , LSTM , Dropout","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:47.358926Z","iopub.execute_input":"2023-03-10T16:58:47.359359Z","iopub.status.idle":"2023-03-10T16:58:56.930829Z","shell.execute_reply.started":"2023-03-10T16:58:47.359302Z","shell.execute_reply":"2023-03-10T16:58:56.929276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled = scaler.transform(X_tr)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:56.933000Z","iopub.execute_input":"2023-03-10T16:58:56.933983Z","iopub.status.idle":"2023-03-10T16:58:56.948368Z","shell.execute_reply.started":"2023-03-10T16:58:56.933942Z","shell.execute_reply":"2023-03-10T16:58:56.946969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled = X_train_scaled.reshape(X_train_scaled.shape[0],X_train_scaled.shape[1],1)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:56.950247Z","iopub.execute_input":"2023-03-10T16:58:56.951349Z","iopub.status.idle":"2023-03-10T16:58:56.960080Z","shell.execute_reply.started":"2023-03-10T16:58:56.951266Z","shell.execute_reply":"2023-03-10T16:58:56.958373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:56.962446Z","iopub.execute_input":"2023-03-10T16:58:56.962959Z","iopub.status.idle":"2023-03-10T16:58:56.975258Z","shell.execute_reply.started":"2023-03-10T16:58:56.962914Z","shell.execute_reply":"2023-03-10T16:58:56.974318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\nmodel.add(GRU(128,return_sequences=True,input_shape=(X_train_scaled.shape[1], X_train_scaled.shape[2])))\nmodel.add(Dropout(0.2))\nmodel.add(GRU(32,return_sequences=True))\nmodel.add(Dropout(0.2))\nmodel.add(GRU(32,return_sequences=True))\nmodel.add(Dropout(0.2))\nmodel.add(GRU(32))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1)) \n\nmodel.compile(optimizer='RMSprop', loss='mae')\nhistory = model.fit(X_train_scaled, \n                    y_tr, \n                    epochs=50,\n                    batch_size=64,\n                    verbose=0)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T16:58:56.976515Z","iopub.execute_input":"2023-03-10T16:58:56.977873Z","iopub.status.idle":"2023-03-10T17:23:01.297330Z","shell.execute_reply.started":"2023-03-10T16:58:56.977832Z","shell.execute_reply":"2023-03-10T17:23:01.295239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate model\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_absolute_percentage_error\n    \ny_val= model.predict(X_train_scaled)\nmae = mean_absolute_error(y_tr, y_val)\nmape=mean_absolute_percentage_error(y_tr, y_val)\nprint('%.5f' % mae)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:01.304926Z","iopub.execute_input":"2023-03-10T17:23:01.305604Z","iopub.status.idle":"2023-03-10T17:23:14.176927Z","shell.execute_reply.started":"2023-03-10T17:23:01.305542Z","shell.execute_reply":"2023-03-10T17:23:14.175488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint('%.5f' % mape)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:14.178820Z","iopub.execute_input":"2023-03-10T17:23:14.183035Z","iopub.status.idle":"2023-03-10T17:23:14.189905Z","shell.execute_reply.started":"2023-03-10T17:23:14.182991Z","shell.execute_reply":"2023-03-10T17:23:14.188485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:14.191911Z","iopub.execute_input":"2023-03-10T17:23:14.192392Z","iopub.status.idle":"2023-03-10T17:23:14.211880Z","shell.execute_reply.started":"2023-03-10T17:23:14.192343Z","shell.execute_reply":"2023-03-10T17:23:14.210427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = X_test_scaled.reshape(X_test_scaled.shape[0],X_test_scaled.shape[1],1)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:14.214016Z","iopub.execute_input":"2023-03-10T17:23:14.214501Z","iopub.status.idle":"2023-03-10T17:23:14.222563Z","shell.execute_reply.started":"2023-03-10T17:23:14.214459Z","shell.execute_reply":"2023-03-10T17:23:14.220999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:14.224974Z","iopub.execute_input":"2023-03-10T17:23:14.225801Z","iopub.status.idle":"2023-03-10T17:23:14.238299Z","shell.execute_reply.started":"2023-03-10T17:23:14.225744Z","shell.execute_reply":"2023-03-10T17:23:14.236921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['time_to_failure'] = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:14.239826Z","iopub.execute_input":"2023-03-10T17:23:14.240272Z","iopub.status.idle":"2023-03-10T17:23:20.933786Z","shell.execute_reply.started":"2023-03-10T17:23:14.240226Z","shell.execute_reply":"2023-03-10T17:23:20.932490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:23:20.935542Z","iopub.execute_input":"2023-03-10T17:23:20.935911Z","iopub.status.idle":"2023-03-10T17:23:20.950216Z","shell.execute_reply.started":"2023-03-10T17:23:20.935878Z","shell.execute_reply":"2023-03-10T17:23:20.948352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:32:47.704347Z","iopub.execute_input":"2023-03-10T17:32:47.705917Z","iopub.status.idle":"2023-03-10T17:32:47.717752Z","shell.execute_reply.started":"2023-03-10T17:32:47.705850Z","shell.execute_reply":"2023-03-10T17:32:47.716460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## BILSTM","metadata":{}},{"cell_type":"markdown","source":"from sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential \nfrom keras.layers import Dense\nfrom keras.layers import GRU\n\nfrom tensorflow.keras.layers import Dense , LSTM , Dropout , Bidirectional\n\nmodel_bi = Sequential()\nmodel_bi.add(Bidirectional(LSTM(128,return_sequences=True),input_shape=(X_train_scaled.shape[1], X_train_scaled.shape[2])))\nmodel.add(Dropout(0.2))\nmodel_bi.add(Bidirectional(LSTM(64)))\nmodel.add(Dropout(0.2))\nmodel_bi.add(Dense(1))\n\n\nmodel_bi.compile(optimizer='RMSprop', loss='mae')\nhistory = model_bi.fit(X_train_scaled, \n                    y_tr, \n                    epochs=50,\n                    batch_size=64,\n                    verbose=1)\n\nmodel_bi.summary()\n\n\n# Evaluate model\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_absolute_percentage_error\n    \ny_val_bilstm= model_bi.predict(X_train_scaled)\nmae_bilstm = mean_absolute_error(y_tr, y_val_bilstm)\nmape_bilstm =mean_absolute_percentage_error(y_tr, y_val_bilstm)\nprint('%.5f' % mae_bilstm)\nprint('%.5f' % mape_bilstm)","metadata":{}}]}