{"cells":[{"metadata":{"_uuid":"598859c3-bdc8-4238-9e33-cf3dfdf743a0","_cell_guid":"98df5d88-c62f-4fb0-a4f8-316b40698ebb","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.svm import NuSVR, SVR\nfrom sklearn.kernel_ridge import KernelRidge\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a6a193e4-be47-4c2e-a9e7-0ef06bfa29da","_cell_guid":"66949c1d-15a0-434d-9aaa-219807e9aaa5","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', nrows=6000000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"acc90eec-0ddf-460e-b9ed-0b1ee6353a56","_cell_guid":"d16a29ce-f593-403d-b770-4d58dde71bb6","trusted":true},"cell_type":"code","source":"train_ad_sample_df = train['acoustic_data'].values[::100]\ntrain_ttf_sample_df = train['time_to_failure'].values[::100]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0caef1f5-50e1-4fec-84ed-3e57d5bcdd79","_cell_guid":"27ff0693-6bd9-4d69-8959-4e5da70d272d","trusted":true},"cell_type":"code","source":"def plot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df, title=\"Acoustic data and time to failure: 1% sampled data\"):\n    fig, ax1 = plt.subplots(figsize=(12, 8))\n    plt.title(title)\n    plt.plot(train_ad_sample_df, color='r')\n    ax1.set_ylabel('acoustic data', color='r')\n    plt.legend(['acoustic data'], loc=(0.01, 0.95))\n    ax2 = ax1.twinx()\n    plt.plot(train_ttf_sample_df, color='b')\n    ax2.set_ylabel('time to failure', color='b')\n    plt.legend(['time to failure'], loc=(0.01, 0.9))\n    plt.grid(True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"997aae5e-0b6b-4469-bef5-aead8c3aad65","_cell_guid":"b2ccdd72-5637-4837-ad41-ac5196cf95e7","trusted":true},"cell_type":"code","source":"plot_acc_ttf_data(train_ad_sample_df, train_ttf_sample_df)\ndel train_ad_sample_df\ndel train_ttf_sample_df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df2cc771-e952-4785-8870-febae5066f16","_cell_guid":"bf563c79-2f9a-412f-a5a3-24ba24c75132","trusted":true},"cell_type":"code","source":"def gen_features(X):\n    strain = []\n    strain.append(X.mean())\n    strain.append(X.std())\n    strain.append(X.min())\n    strain.append(X.max())\n    strain.append(X.kurtosis())\n    strain.append(X.skew())\n    strain.append(np.quantile(X,0.01))\n    strain.append(np.quantile(X,0.05))\n    strain.append(np.quantile(X,0.95))\n    strain.append(np.quantile(X,0.99))\n    strain.append(np.abs(X).max())\n    strain.append(np.abs(X).mean())\n    strain.append(np.abs(X).std())\n    return pd.Series(strain)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a21eeb4c-8af7-48a8-9930-26836e1aa5c7","_cell_guid":"b119460b-3408-4a97-891a-ad92d4813d76","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', iterator=True, chunksize=150_000, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bd39b10b-9508-4dc6-90b5-b689b9554fef","_cell_guid":"a7a46b4c-410b-40f8-8ca3-75f3d807b601","trusted":true},"cell_type":"code","source":"X_train = pd.DataFrame()\ny_train = pd.Series()\nfor df in train:\n    ch = gen_features(df['acoustic_data'])\n    X_train = X_train.append(ch, ignore_index=True)\n    y_train = y_train.append(pd.Series(df['time_to_failure'].values[-1]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0b18fc29-e76c-403d-bbf3-6e324c3ae720","_cell_guid":"1c152a56-de4d-4c01-941e-84b44da1e682","trusted":true},"cell_type":"code","source":"X_train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a6f982a9-5dec-4fe5-8c82-1f95e84ffe4c","_cell_guid":"93905913-876d-4dc6-b5a8-e1e8f7d6c512","trusted":true},"cell_type":"code","source":"train_pool = Pool(X_train, y_train)\nm = CatBoostRegressor(iterations=10000, loss_function='MAE', boosting_type='Ordered')\nm.fit(X_train, y_train, silent=True)\nm.best_score_","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5756793c-b222-4f1f-a1bb-2e0fa8a1d93d","_cell_guid":"d8cbef61-2b72-40d6-8595-e89996ae9b4b","trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.svm import NuSVR, SVR\n\n\nscaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\n\nparameters = [{'gamma': [0.001, 0.005, 0.01, 0.02, 0.05, 0.1],\n               'C': [0.1, 0.2, 0.25, 0.5, 1, 1.5, 2]}]\n               #'nu': [0.75, 0.8, 0.85, 0.9, 0.95, 0.97]}]\n\nreg1 = GridSearchCV(SVR(kernel='rbf', tol=0.01), parameters, cv=5, scoring='neg_mean_absolute_error')\nreg1.fit(X_train_scaled, y_train.values.flatten())\ny_pred1 = reg1.predict(X_train_scaled)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"78253cf4-a182-4ac3-a52f-6ba5ba282b4b","_cell_guid":"cfb50973-47a6-48c4-90e2-d40392773e15","trusted":true},"cell_type":"code","source":"print(\"Best CV score: {:.4f}\".format(reg1.best_score_))\nprint(reg1.best_params_)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0d47c8f7-07c5-4629-9bf4-fec6113741dc","_cell_guid":"787af9b2-916d-4832-9538-c7022b59b8dc","trusted":true},"cell_type":"code","source":"from sklearn.kernel_ridge import KernelRidge\n\nparameters = [{'gamma': np.linspace(0.001, 0.1, 10),\n               'alpha': [0.005, 0.01, 0.02, 0.05, 0.1]}]\n\nreg2 = GridSearchCV(KernelRidge(kernel='rbf'), parameters, cv=5, scoring='neg_mean_absolute_error')\nreg2.fit(X_train_scaled, y_train.values.flatten())\ny_pred2 = reg2.predict(X_train_scaled)\n\nprint(\"Best CV score: {:.4f}\".format(reg2.best_score_))\nprint(reg2.best_params_)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7752c3e0-0b37-4cb0-83b3-ed77d89b96b7","_cell_guid":"88a8fa8c-6c60-4389-80dd-8466348998e9","trusted":true},"cell_type":"code","source":"import lightgbm as lgb\nfrom tqdm import tqdm_notebook as tqdm\nimport random\n\n\nfixed_params = {\n    'objective': 'regression_l1',\n    'boosting': 'gbdt',\n    'verbosity': -1,\n    'random_seed': 19\n}\n\nparam_grid = {\n    'learning_rate': [0.1, 0.08, 0.05, 0.01],\n    'num_leaves': [32, 46, 52, 58, 68, 72, 80, 92],\n    'max_depth': [3, 4, 5, 6, 8, 12, 16, -1],\n    'feature_fraction': [0.8, 0.85, 0.9, 0.95, 1],\n    'subsample': [0.8, 0.85, 0.9, 0.95, 1],\n    'lambda_l1': [0, 0.1, 0.2, 0.4, 0.6, 0.9],\n    'lambda_l2': [0, 0.1, 0.2, 0.4, 0.6, 0.9],\n    'min_data_in_leaf': [10, 20, 40, 60, 100],\n    'min_gain_to_split': [0, 0.001, 0.01, 0.1],\n}\n\nbest_score = 999\ndataset = lgb.Dataset(X_train, label=y_train)  # no need to scale features\n\nfor i in tqdm(range(200)):\n    params = {k: random.choice(v) for k, v in param_grid.items()}\n    params.update(fixed_params)\n    result = lgb.cv(params, dataset, nfold=5, early_stopping_rounds=50,\n                    num_boost_round=20000, stratified=False)\n    \n    if result['l1-mean'][-1] < best_score:\n        best_score = result['l1-mean'][-1]\n        best_params = params\n        best_nrounds = len(result['l1-mean'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c86b3ac6-f8bd-4df6-b66b-b711c4a55522","_cell_guid":"2573e940-0808-4190-b458-333306e097df","trusted":true},"cell_type":"code","source":"print(\"Best mean score: {:.4f}, num rounds: {}\".format(best_score, best_nrounds))\nprint(best_params)\nreg3 = lgb.train(best_params, dataset, best_nrounds)\ny_pred3 = reg3.predict(X_train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"395c57a8-5ff7-42ba-b8d5-4648063f004f","_cell_guid":"a32d63fc-971d-4c15-8fc4-3fff0fa5d2fd","trusted":true},"cell_type":"code","source":"plt.tight_layout()\nf = plt.figure(figsize=(12, 6))\nf.add_subplot(1, 3, 1)\nplt.scatter(y_train.values.flatten(), y_pred1)\nplt.title('SVR', fontsize=16)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.ylabel('predicted', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\n\nf.add_subplot(1, 3, 2)\nplt.scatter(y_train.values.flatten(), y_pred2)\nplt.title('Kernel Ridge', fontsize=16)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\n\nf.add_subplot(1, 3, 3)\nplt.scatter(y_train.values.flatten(), y_pred3)\nplt.title('Gradient boosting', fontsize=16)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\nplt.show(block=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ee1f0da-8778-4493-8d95-9a1afa143e5f","_cell_guid":"cd0321ed-5ab1-4876-8842-622ae2b3c4f7","trusted":true},"cell_type":"code","source":"# second plot\nplt.figure(figsize=(10, 5))\nplt.plot(y_train.values.flatten(), color='blue', label='y_train')\nplt.plot(y_pred1, color='orange', label='SVR')\nplt.legend()\nplt.title('SVR predictions vs actual')\n\n# third plot\nplt.figure(figsize=(10, 5))\nplt.plot(y_train.values.flatten(), color='blue', label='y_train')\nplt.plot(y_pred2, color='gray', label='KernelRidge')\nplt.legend()\nplt.title('Kernel Ridge predictions vs actual')\n\n# fourth plot\nplt.figure(figsize=(10, 5))\nplt.plot(y_train.values.flatten(), color='blue', label='y_train')\nplt.plot(y_pred3, color='green', label='Gradient boosting')\nplt.legend()\nplt.title('GBDT predictions vs actual')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"120ca93f-a564-4054-81d1-d3a385118b15","_cell_guid":"445c1969-8444-461a-a65a-37ccee6af87b","trusted":true},"cell_type":"code","source":"# DISCLAIMER: THIS CODE IS A RESULT OF FOLLOWING THIS TUTORIAL: https://www.youtube.com/watch?v=TffGdSsWKlA","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}