{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:17:37.106722Z","iopub.execute_input":"2022-05-22T11:17:37.10748Z","iopub.status.idle":"2022-05-22T11:17:37.112899Z","shell.execute_reply.started":"2022-05-22T11:17:37.107427Z","shell.execute_reply":"2022-05-22T11:17:37.111764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import mean_absolute_error, max_error, mean_absolute_percentage_error\nfrom scipy.stats import kurtosis, skew\nimport shap","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:17:37.15714Z","iopub.execute_input":"2022-05-22T11:17:37.157421Z","iopub.status.idle":"2022-05-22T11:17:41.381038Z","shell.execute_reply.started":"2022-05-22T11:17:37.157391Z","shell.execute_reply":"2022-05-22T11:17:41.37973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\npd.options.display.precision = 15\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:17:41.383114Z","iopub.execute_input":"2022-05-22T11:17:41.383386Z","iopub.status.idle":"2022-05-22T11:21:53.364233Z","shell.execute_reply.started":"2022-05-22T11:17:41.383354Z","shell.execute_reply":"2022-05-22T11:21:53.362513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset in input is divided in segments of 150 000 samples each (as is divided the test set for the challange), it generates 4194 segments","metadata":{}},{"cell_type":"code","source":"rows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64, columns=['ave','std','max','min','skew','kurtosis'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64, columns=['time_to_failure'])\n\nfor segment in range(segments):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    y_train.loc[segment, 'time_to_failure'] = y\n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()\n    X_train.loc[segment, 'skew'] = skew(x)\n    X_train.loc[segment, 'kurtosis'] = kurtosis(x)    ","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:21:53.366631Z","iopub.execute_input":"2022-05-22T11:21:53.367948Z","iopub.status.idle":"2022-05-22T11:22:16.540042Z","shell.execute_reply.started":"2022-05-22T11:21:53.367877Z","shell.execute_reply":"2022-05-22T11:22:16.538746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:27.873426Z","iopub.execute_input":"2022-05-22T11:25:27.873735Z","iopub.status.idle":"2022-05-22T11:25:27.889849Z","shell.execute_reply.started":"2022-05-22T11:25:27.873708Z","shell.execute_reply":"2022-05-22T11:25:27.888937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then normalize the training data","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\ny_train_flatten = y_train.values.flatten()","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:30.717465Z","iopub.execute_input":"2022-05-22T11:25:30.718205Z","iopub.status.idle":"2022-05-22T11:25:30.729157Z","shell.execute_reply.started":"2022-05-22T11:25:30.718147Z","shell.execute_reply":"2022-05-22T11:25:30.727993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot(y_train_flatten, y_pred):\n    plt.figure(figsize=(6, 6))\n    plt.scatter(y_train_flatten, y_pred)\n    plt.xlim(0, 20)\n    plt.ylim(0, 20)\n    plt.xlabel('actual', fontsize=12)\n    plt.ylabel('predicted', fontsize=12)\n    plt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:33.410446Z","iopub.execute_input":"2022-05-22T11:25:33.411012Z","iopub.status.idle":"2022-05-22T11:25:33.419118Z","shell.execute_reply.started":"2022-05-22T11:25:33.410953Z","shell.execute_reply":"2022-05-22T11:25:33.417839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def score(y_train_flatten, y_pred):\n    max = max_error(y_train_flatten, y_pred)\n    mae = mean_absolute_error(y_train_flatten, y_pred)\n    mape = mean_absolute_percentage_error(y_train_flatten, y_pred)\n    print(f'Max Error: {max:0.3f}')\n    print(f'Mean Absolute Error: {mae:0.3f}')\n    print(f'Mean Absolute Percentage Error: {mape:0.3f}')    ","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:35.737827Z","iopub.execute_input":"2022-05-22T11:25:35.738121Z","iopub.status.idle":"2022-05-22T11:25:35.747411Z","shell.execute_reply.started":"2022-05-22T11:25:35.738089Z","shell.execute_reply":"2022-05-22T11:25:35.746195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Catboost with Root Mean Square Error","metadata":{}},{"cell_type":"code","source":"m_rmse = CatBoostRegressor()\nm_rmse.fit(X_train_scaled, y_train.values.flatten(), silent=True)\ny_pred_m_rmse = m_rmse.predict(X_train_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:37.510795Z","iopub.execute_input":"2022-05-22T11:25:37.512028Z","iopub.status.idle":"2022-05-22T11:25:39.978709Z","shell.execute_reply.started":"2022-05-22T11:25:37.511978Z","shell.execute_reply":"2022-05-22T11:25:39.977583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot(y_train_flatten, y_pred_m_rmse)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:43.904417Z","iopub.execute_input":"2022-05-22T11:25:43.906141Z","iopub.status.idle":"2022-05-22T11:25:44.23816Z","shell.execute_reply.started":"2022-05-22T11:25:43.906071Z","shell.execute_reply":"2022-05-22T11:25:44.236885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score(y_train_flatten, y_pred_m_rmse)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:47.041295Z","iopub.execute_input":"2022-05-22T11:25:47.041655Z","iopub.status.idle":"2022-05-22T11:25:47.050579Z","shell.execute_reply.started":"2022-05-22T11:25:47.041624Z","shell.execute_reply":"2022-05-22T11:25:47.049464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Interpretation of the model","metadata":{}},{"cell_type":"code","source":"explainer = shap.Explainer(m_rmse)\nshap_values = explainer(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:50.419887Z","iopub.execute_input":"2022-05-22T11:25:50.420434Z","iopub.status.idle":"2022-05-22T11:25:51.247986Z","shell.execute_reply.started":"2022-05-22T11:25:50.420392Z","shell.execute_reply":"2022-05-22T11:25:51.246532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.plots.bar(shap_values)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:25:53.56422Z","iopub.execute_input":"2022-05-22T11:25:53.564595Z","iopub.status.idle":"2022-05-22T11:25:53.803306Z","shell.execute_reply.started":"2022-05-22T11:25:53.564562Z","shell.execute_reply":"2022-05-22T11:25:53.802223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot above shows the **importance** of the feature in the feature in scoring a segment.","metadata":{}},{"cell_type":"code","source":"shap.plots.beeswarm(shap_values)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:26:15.910264Z","iopub.execute_input":"2022-05-22T11:26:15.911819Z","iopub.status.idle":"2022-05-22T11:26:16.514735Z","shell.execute_reply.started":"2022-05-22T11:26:15.911723Z","shell.execute_reply":"2022-05-22T11:26:16.513235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot above shows how the calculated features in the dataset impact the model’s output.\nThe blue points are associated with low values, the red ones with high values.","metadata":{}},{"cell_type":"code","source":"shap.initjs()\nshap.plots.force(shap_values[1651])","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:26:21.624941Z","iopub.execute_input":"2022-05-22T11:26:21.625319Z","iopub.status.idle":"2022-05-22T11:26:21.664381Z","shell.execute_reply.started":"2022-05-22T11:26:21.625283Z","shell.execute_reply":"2022-05-22T11:26:21.663634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The graph above shows how the different features impact the score calculated for the segment 1651. The mangitude of the arrows corresponds to the weight of the features shown in the first graph of this section.","metadata":{}},{"cell_type":"markdown","source":"# Submission of the model","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/LANL-Earthquake-Prediction/sample_submission.csv', index_col='seg_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:26:40.493187Z","iopub.execute_input":"2022-05-22T11:26:40.493501Z","iopub.status.idle":"2022-05-22T11:26:40.52488Z","shell.execute_reply.started":"2022-05-22T11:26:40.493471Z","shell.execute_reply":"2022-05-22T11:26:40.524202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:27:16.552068Z","iopub.execute_input":"2022-05-22T11:27:16.552462Z","iopub.status.idle":"2022-05-22T11:27:16.560532Z","shell.execute_reply.started":"2022-05-22T11:27:16.552426Z","shell.execute_reply":"2022-05-22T11:27:16.559543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for segment_id in X_test.index:\n    filename = \"../input/LANL-Earthquake-Prediction/test/{0}.csv\".format(segment_id)\n    segment =  pd.read_csv(filename)\n    x = segment['acoustic_data'].values\n    X_test.loc[segment_id, 'ave'] = x.mean()\n    X_test.loc[segment_id, 'std'] = x.std()\n    X_test.loc[segment_id, 'max'] = x.max()\n    X_test.loc[segment_id, 'min'] = x.min()\n    X_test.loc[segment_id, 'skew'] = skew(x)\n    X_test.loc[segment_id, 'kurtosis'] = kurtosis(x)    ","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:27:31.494719Z","iopub.execute_input":"2022-05-22T11:27:31.495213Z","iopub.status.idle":"2022-05-22T11:29:05.782726Z","shell.execute_reply.started":"2022-05-22T11:27:31.495181Z","shell.execute_reply":"2022-05-22T11:29:05.781106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = m_rmse.predict(X_test_scaled)\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-22T11:30:10.261051Z","iopub.execute_input":"2022-05-22T11:30:10.261705Z","iopub.status.idle":"2022-05-22T11:30:10.301934Z","shell.execute_reply.started":"2022-05-22T11:30:10.261665Z","shell.execute_reply":"2022-05-22T11:30:10.300895Z"},"trusted":true},"execution_count":null,"outputs":[]}]}