{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Introduction\nJoost Jaspers (<student_name>), <Kaggle username>; \\\nJan-Paul Metz (5327253), <Kaggle username>. \n\nScore: \\\nLeaderboard rank:","metadata":{}},{"cell_type":"markdown","source":"# 2.Data","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-23T19:55:53.096586Z","iopub.execute_input":"2023-04-23T19:55:53.097456Z","iopub.status.idle":"2023-04-23T19:55:53.102949Z","shell.execute_reply.started":"2023-04-23T19:55:53.097391Z","shell.execute_reply":"2023-04-23T19:55:53.101580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1 Dataset\nIn this section we will load and explore the dataset. ","metadata":{}},{"cell_type":"code","source":"# Loading data \ntrain_df = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/train.csv', nrows=200000000, \n                    dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n\ntrain_df.head(5)\nX_plot = train_df[\"acoustic_data\"].values[::100]\ny_plot = train_df[\"time_to_failure\"].values[::100]","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:55:53.126228Z","iopub.execute_input":"2023-04-23T19:55:53.126839Z","iopub.status.idle":"2023-04-23T19:56:56.286210Z","shell.execute_reply.started":"2023-04-23T19:55:53.126797Z","shell.execute_reply":"2023-04-23T19:56:56.284689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot a segment of train_df \ntrain_df_acoustic_segment = train_df[\"acoustic_data\"].values[::1000]\ntrain_df_time_to_failure_segment = train_df[\"time_to_failure\"].values[::1000]\n\nfig, ax1 = plt.subplots(figsize=(20,10))\nplt.plot(train_df_acoustic_segment, color=\"r\")\nax1.set_ylabel('acoustic data', color=\"r\")\nplt.legend(['acoustic data'], loc=(0.9,0.93))\n\nax2 = ax1.twinx()\nplt.plot(train_df_time_to_failure_segment, color=\"b\")\nax2.set_ylabel('time to failure', color=\"b\")\nplt.legend(['time to failure'], loc=(0.9,0.9))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:56:56.288565Z","iopub.execute_input":"2023-04-23T19:56:56.288893Z","iopub.status.idle":"2023-04-23T19:56:56.992873Z","shell.execute_reply.started":"2023-04-23T19:56:56.288858Z","shell.execute_reply":"2023-04-23T19:56:56.991579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the figure above you can see that just before a failure the acoustic data shows a spike. Also in the middle a peak shows up, a rather small peak. For training the data we can use simple features as mean and variance. ","metadata":{}},{"cell_type":"markdown","source":"# 2.2 Data Exploration\nFor this problem we need to segment the dataset in order to get usable features. In order to predict the time to failure we want to know information about a relative short time ago. From this data segments we can extract feratures as mean and variance. ","metadata":{}},{"cell_type":"code","source":"# X and y: \nX = train_df[\"acoustic_data\"].values\ny = train_df[\"time_to_failure\"].values","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:56:57.013925Z","iopub.execute_input":"2023-04-23T19:56:57.014720Z","iopub.status.idle":"2023-04-23T19:56:57.020910Z","shell.execute_reply.started":"2023-04-23T19:56:57.014676Z","shell.execute_reply":"2023-04-23T19:56:57.019690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data in segments \ndef segment(X, rows):    \n    # Make row size fit for size (rows,n_segments)\n    n_segments = np.floor(len(X)/rows)\n    X_segments = np.zeros(shape=(int(rows),int(n_segments)))    \n    for i in range(int(n_segments)):\n        X_segments[:,i] =  X[int(i*rows): int(i*rows+rows)] \n        \n    X_segments = np.array(X_segments)\n    return X_segments \n\n\n# Take segments where number of rows is number of samples: \nrows = 99_999 # n_samples\nX_seg = segment(X, rows)\ny_seg = segment(y, rows)\nprint(np.max(y),np.max(y_seg))\nprint(\"X shape: \", X_seg.shape, \"y shape: \", y_seg.shape)\n# Pick last value of each segment\ny_seg = y_seg[-1,:]\nprint(\"y shape:\", y_seg.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:56:57.030244Z","iopub.execute_input":"2023-04-23T19:56:57.030700Z","iopub.status.idle":"2023-04-23T19:57:17.548492Z","shell.execute_reply.started":"2023-04-23T19:56:57.030650Z","shell.execute_reply":"2023-04-23T19:57:17.546883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_seg)\n# print(y_seg_last.shape)\n# print(np.min(y_seg_last))\n# print(np.max(y_seg_last))","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:17.550142Z","iopub.execute_input":"2023-04-23T19:57:17.550689Z","iopub.status.idle":"2023-04-23T19:57:17.557457Z","shell.execute_reply.started":"2023-04-23T19:57:17.550642Z","shell.execute_reply":"2023-04-23T19:57:17.555494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract features from X_seg\n# x is of shape (rows, n_segments)\ndef extract_features(x):\n    n_features = 3\n    X_mean = np.mean(x,axis=0)\n    X_std = np.std(x,axis=0)\n    X_abs_max = np.max(abs(x),axis=0)\n    X_features = np.zeros(shape=(x.shape[0],n_features))\n    X_features = [X_mean, X_std, X_abs_max]\n    return np.array(X_features).T\n\nX_features = extract_features(X_seg)\nprint(X_features.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:17.559351Z","iopub.execute_input":"2023-04-23T19:57:17.559856Z","iopub.status.idle":"2023-04-23T19:57:19.745288Z","shell.execute_reply.started":"2023-04-23T19:57:17.559791Z","shell.execute_reply":"2023-04-23T19:57:19.743782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we want to plot the features for inspection. ","metadata":{}},{"cell_type":"code","source":"print(X_features[:,0])\nplt.figure()\nplt.plot(X_features[:,0], X_features[:,1], 'o')\nplt.figure()\nplt.plot(X_features[:,0], X_features[:,2], 'o')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:19.747148Z","iopub.execute_input":"2023-04-23T19:57:19.748369Z","iopub.status.idle":"2023-04-23T19:57:20.218676Z","shell.execute_reply.started":"2023-04-23T19:57:19.748314Z","shell.execute_reply":"2023-04-23T19:57:20.217288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.3 Data preperation\nBefore we start training, we need to normalize the data. There are no 'gaps' or other missing data, so we do not need to improve the data further.","metadata":{}},{"cell_type":"code","source":"# Normalize the data: \nfrom sklearn import preprocessing \nscaler = preprocessing.StandardScaler().fit(X_features)\nX_features = scaler.transform(X_features)\nprint(\"shape X_features\", X_features.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:20.220087Z","iopub.execute_input":"2023-04-23T19:57:20.221004Z","iopub.status.idle":"2023-04-23T19:57:20.229535Z","shell.execute_reply.started":"2023-04-23T19:57:20.220961Z","shell.execute_reply":"2023-04-23T19:57:20.228084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1.1 Train-test split\nData is split in test and train sets.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Split data in train test data\nX_train, X_test, y_train, y_test = train_test_split(X_features, y_seg, test_size=0.2, shuffle=True, random_state=42)\nprint(\"shape X_train\", X_train.shape)\nprint(np.max(y_seg))","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:20.231035Z","iopub.execute_input":"2023-04-23T19:57:20.231391Z","iopub.status.idle":"2023-04-23T19:57:20.242162Z","shell.execute_reply.started":"2023-04-23T19:57:20.231357Z","shell.execute_reply":"2023-04-23T19:57:20.240773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and Results \n\nFirst we define a model based on logistic regression. Finally we will fit the model to the data of y_train","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\n\n# Define some models \nlin_reg = LinearRegression()\ntree_reg = DecisionTreeRegressor(random_state=0)\nforest_reg = RandomForestRegressor(random_state=0, max_depth=6 )\n\n# Fit models\nlin_reg.fit(X_train, y_train)\ntree_reg.fit(X_train, y_train)\nforest_reg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:20.244024Z","iopub.execute_input":"2023-04-23T19:57:20.244715Z","iopub.status.idle":"2023-04-23T19:57:20.546982Z","shell.execute_reply.started":"2023-04-23T19:57:20.244664Z","shell.execute_reply":"2023-04-23T19:57:20.545813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For testing this model, testdata is needed. We take the same features from the test set and test by getting the score.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score \nfrom sklearn.metrics import mean_squared_error\nprint(\"y_test shape:\", y_test.shape)\n\nscore_lin = lin_reg.score(X_test, y_test)\nscore_tree = tree_reg.score(X_test,y_test)\nscore_forest = forest_reg.score(X_test,y_test)\n\nprint(\"lin reg score: \", score_lin, \"tree reg score: \", score_tree)\nprint(\"forest score: \", score_forest)\npred = forest_reg.predict(X_test)\nplt.figure()\nplt.plot(y_test, '+', color='b')\nplt.plot(pred, 'o', color='r')\nprint(\"pred shape:\", pred.shape)\ncross_val_score(tree_reg, X_test, y_test, cv=10)\nprint(np.max(y))","metadata":{"execution":{"iopub.status.busy":"2023-04-23T19:57:20.548547Z","iopub.execute_input":"2023-04-23T19:57:20.548878Z","iopub.status.idle":"2023-04-23T19:57:21.002878Z","shell.execute_reply.started":"2023-04-23T19:57:20.548849Z","shell.execute_reply":"2023-04-23T19:57:21.001628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}