{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-05T16:32:00.758065Z","iopub.execute_input":"2023-11-05T16:32:00.758684Z","iopub.status.idle":"2023-11-05T16:32:01.217251Z","shell.execute_reply.started":"2023-11-05T16:32:00.758642Z","shell.execute_reply":"2023-11-05T16:32:01.216031Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.linear_model import LinearRegression\ndef fea_eng(df):\n    return (df.assign(std_nopeak_next=lambda df:df['std_nopeak']-df['std_nopeak'].shift(1),\n         std_nopeak_prev=lambda df:df['std_nopeak']-df['std_nopeak'].shift(-1),\n        iqr_next=lambda df:df['iqr']-df['iqr'].shift(1),\n        iqr_prev=lambda df:df['iqr']-df['iqr'].shift(-1),))\n\ndef linear_regression(X, y):\n#     # Convert the input columns to numpy arrays\n    X = np.array(X).reshape(-1, 1)\n    y = np.array(y)\n\n    # Create an instance of the LinearRegression model\n    model = LinearRegression()\n\n    # Fit the model to the data\n    model.fit(X, y)\n\n    # Return the trained model\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-11-05T16:32:01.221488Z","iopub.execute_input":"2023-11-05T16:32:01.221987Z","iopub.status.idle":"2023-11-05T16:32:03.410433Z","shell.execute_reply.started":"2023-11-05T16:32:01.22195Z","shell.execute_reply":"2023-11-05T16:32:03.409139Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Load the data\ntrain_y = pd.read_parquet('/kaggle/input/eda-earthquake-prediction/train.parquet.gzip', columns=['time_stamp','ttf', 'EP_disp'])\nprint(train_y.shape)\ntest_y = pd.read_parquet('/kaggle/input/eda-earthquake-prediction/test.parquet.gzip', columns=['time_stamp'])\nprint(test_y.shape)\ntrain_X = pd.read_parquet('/kaggle/input/20231023earthquake/train_feaeng.parquet.gzip')\ntest_X = pd.read_parquet('/kaggle/input/20231023earthquake/test_feaeng.parquet.gzip')\ntrain_X=train_X.pipe(fea_eng)\ntest_X=test_X.pipe(fea_eng)\n# prepare data to train RandomForestClassifier to predict when ttf==0\ncol_feaeng=['mean', 'iqr', 'iqr2', 'iqr3', 'iqr4', 'std', 'std_nopeak']+['std_nopeak_next','std_nopeak_prev','iqr_next','iqr_prev']\nttf0=(train_X[col_feaeng].query(\"std_nopeak>9\")\n    .join(train_y.query(\"ttf==0\")[['ttf']])\n    .assign(ttf0=lambda df:(df['ttf']==0).astype(int))\n    .drop('ttf', axis=1)\n)\n# RandomForestClassifier to predict when ttf==0\nmodel_ttf0 = RandomForestClassifier(max_depth=3, n_estimators=100, random_state=20)\n# Train the model on the training set\nmodel_ttf0.fit(ttf0[col_feaeng], ttf0['ttf0'])\n# Make predictions on the training & validation set\nttf0_tr_pred = model_ttf0.predict(ttf0[col_feaeng]).round(1)\nprint(confusion_matrix(ttf0['ttf0'], ttf0_tr_pred))\ntest_X_ttf0=(test_X[col_feaeng].query(\"std_nopeak>9\").assign(ttf0=lambda df:model_ttf0.predict(df[col_feaeng]).round(1))\n.query(\"ttf0==1\")\n)\n# modify to auto from manually adjust\nindex_drop=list(test_X_ttf0.reset_index()\n    .assign(index_diff=lambda df:df['index'].diff())\n    .query(\"index_diff<100\")['index'].unique()\n    )\nprint(f\"drop index {index_drop} since peaks are too close to be true\")\ntest_X_ttf0=test_X_ttf0.drop(index_drop)[['ttf0']]","metadata":{"execution":{"iopub.status.busy":"2023-11-05T16:32:03.415752Z","iopub.execute_input":"2023-11-05T16:32:03.416194Z","iopub.status.idle":"2023-11-05T16:32:05.218038Z","shell.execute_reply.started":"2023-11-05T16:32:03.416148Z","shell.execute_reply":"2023-11-05T16:32:05.217062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ttf0=ttf0.query(\"ttf0==1\").join(train_y).iloc[1:]\n# linear_regression to predict EP_disp when ttf==0\nmodel_EP_disp=linear_regression(ttf0['time_stamp'], ttf0['EP_disp'])\nttf_pred=(\n    test_X_ttf0\n.join(test_y)\n)\nttf_pred=ttf_pred.assign(EP_disp=model_EP_disp.predict(ttf_pred[['time_stamp']].values))[['time_stamp','EP_disp']]\nextrapol=ttf_pred.diff().median()+ttf_pred.tail(1)\nttf_pred4=(test_y.join(ttf_pred.add_suffix('_'))\n        .assign(time_stamp_=lambda df:df['time_stamp_'].bfill(),\n                EP_disp_=lambda df:df['EP_disp_'].bfill(),\n                )\n            .fillna({'time_stamp_':extrapol['time_stamp'].values[0],\n                    'EP_disp_':extrapol['EP_disp'].values[0],\n                    })\n            .assign(ttf=lambda df:df['time_stamp_']-df['time_stamp'], EP_disp=lambda df:df['EP_disp_'])\n            .drop(columns=['time_stamp_','EP_disp_'])\n            .iloc[4:,:]\n            .reset_index(drop=True)\n) \n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-05T16:32:05.220238Z","iopub.execute_input":"2023-11-05T16:32:05.220809Z","iopub.status.idle":"2023-11-05T16:32:05.304464Z","shell.execute_reply.started":"2023-11-05T16:32:05.220759Z","shell.execute_reply":"2023-11-05T16:32:05.302745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions=(test_y.iloc[4:,:].reset_index().assign(index=lambda df:df['index']-3).reset_index(drop=True)\n            .assign(ttf=ttf_pred4['ttf'], EP_disp=ttf_pred4['EP_disp'])\n            .assign(EP_disp_FUT_1=lambda df:df['EP_disp'].fillna(-1)) \n            .assign(EP_disp_FUT_2=lambda df:df['EP_disp'].fillna(-1)) \n            .assign(EP_disp_FUT_3=lambda df:df['EP_disp'].fillna(-1)) \n            )\npredictions.to_csv('/kaggle/working/predictions.csv', index=False)\npredictions.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-05T16:32:05.306174Z","iopub.execute_input":"2023-11-05T16:32:05.30651Z","iopub.status.idle":"2023-11-05T16:32:07.709016Z","shell.execute_reply.started":"2023-11-05T16:32:05.306482Z","shell.execute_reply":"2023-11-05T16:32:07.707568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n# Save the model to a file\njoblib.dump(model_ttf0, '/kaggle/working/model_ttf0.pkl')\njoblib.dump(model_EP_disp, '/kaggle/working/model_EP_disp.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-11-05T16:32:07.710685Z","iopub.execute_input":"2023-11-05T16:32:07.711115Z","iopub.status.idle":"2023-11-05T16:32:07.803739Z","shell.execute_reply.started":"2023-11-05T16:32:07.711082Z","shell.execute_reply":"2023-11-05T16:32:07.802361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# not run below","metadata":{}}]}