{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11000,"databundleVersionId":875412,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Introduction\nSeppe Dijkstra:\n* username: Seppe Dijkstra\n\nJurre Botman:\n* username: Jurre Botman\n","metadata":{}},{"cell_type":"code","source":"import sklearn\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib import pyplot as plt\nimport pandas as pd \nimport numpy as np\nfrom scipy.io import wavfile\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestRegressor \nfrom sklearn.decomposition import PCA\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:44:07.810111Z","iopub.execute_input":"2024-04-10T12:44:07.810671Z","iopub.status.idle":"2024-04-10T12:44:07.819089Z","shell.execute_reply.started":"2024-04-10T12:44:07.810634Z","shell.execute_reply":"2024-04-10T12:44:07.817242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Data","metadata":{}},{"cell_type":"markdown","source":"**2.1 Dataset**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing the data**\n","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('../input/LANL-Earthquake-Prediction/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure':np.float64}, nrows=300000000)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:10:32.164122Z","iopub.execute_input":"2024-04-10T12:10:32.165168Z","iopub.status.idle":"2024-04-10T12:12:35.507012Z","shell.execute_reply.started":"2024-04-10T12:10:32.165123Z","shell.execute_reply":"2024-04-10T12:12:35.505305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Splitting the data into train and test data**","metadata":{}},{"cell_type":"code","source":"x_data = data['acoustic_data'].values[:300000000] \ny_data = data['time_to_failure'].values[:300000000] \n\ndata_train,data_test = train_test_split(data,train_size=0.7, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:26:09.092666Z","iopub.execute_input":"2024-04-10T12:26:09.093388Z","iopub.status.idle":"2024-04-10T12:26:18.572255Z","shell.execute_reply.started":"2024-04-10T12:26:09.093275Z","shell.execute_reply":"2024-04-10T12:26:18.570849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Exploring the data**","metadata":{}},{"cell_type":"code","source":"acoustic_data = x_data[::100]\ntime_to_failure_data = y_data[::100]\n\nfig, ax1 = plt.subplots(figsize=(12, 8))\nplt.title(\"Time to failure and Acoustic data from 500 000 000 samples\")\nplt.plot(acoustic_data, color='g')\nax1.set_ylabel('acoustic data', color='g')\nplt.legend(['acoustic data'])\nax2 = ax1.twinx()\nplt.plot(time_to_failure_data, color='r')\nax2.set_ylabel('time to failure', color='r')\nplt.legend(['time to failure'])\nplt.grid()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:26:20.771481Z","iopub.execute_input":"2024-04-10T12:26:20.772036Z","iopub.status.idle":"2024-04-10T12:26:27.143630Z","shell.execute_reply.started":"2024-04-10T12:26:20.772000Z","shell.execute_reply":"2024-04-10T12:26:27.141648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extracting the features from the data**","metadata":{}},{"cell_type":"code","source":"rows = 150_000\nsegments = int(np.floor(data_train.shape[0] / rows))","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:32:19.865338Z","iopub.execute_input":"2024-04-10T12:32:19.866445Z","iopub.status.idle":"2024-04-10T12:32:19.880894Z","shell.execute_reply.started":"2024-04-10T12:32:19.866402Z","shell.execute_reply":"2024-04-10T12:32:19.879253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_features_from_segment(segment, seg, X):\n    x = pd.Series(seg['acoustic_data'].values)\n\n    X.loc[segment, 'mean'] = x.mean()\n    X.loc[segment, 'std'] = x.std()\n    X.loc[segment, 'max'] = x.max()\n    X.loc[segment, 'min'] = x.min()\n    X.loc[segment, 'kurt'] = x.kurtosis()\n    X.loc[segment, 'skew'] = x.skew()\n    \n#Rolling window of size 50\n    x_roll_std = x.rolling(window = 50).std().dropna().values\n    x_roll_mean = x.rolling(window= 50).mean().dropna().values\n    \n    #Features from rolling mean\n    X.loc[segment, 'rolling_mean'] =x_roll_mean.mean()\n    X.loc[segment, 'rolling_mean'] =x_roll_mean.std()\n    \n    #Features from rolling std\n    X.loc[segment, 'rolling_std'] = x_roll_std.mean()\n    X.loc[segment, 'rolling_std'] = x_roll_std.std()","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:14:59.375698Z","iopub.execute_input":"2024-04-10T13:14:59.376148Z","iopub.status.idle":"2024-04-10T13:14:59.386821Z","shell.execute_reply.started":"2024-04-10T13:14:59.376117Z","shell.execute_reply":"2024-04-10T13:14:59.385351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64)\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64, columns=['time_to_failure'])\nfor segment in range(segments):\n    seg = data_train.iloc[segment*rows:segment*rows+rows]\n    extract_features_from_segment(segment, seg, X_train)\n    y_train.loc[segment, 'time_to_failure'] = seg['time_to_failure'].values[-1]","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:15:01.980746Z","iopub.execute_input":"2024-04-10T13:15:01.981712Z","iopub.status.idle":"2024-04-10T13:15:26.873089Z","shell.execute_reply.started":"2024-04-10T13:15:01.981667Z","shell.execute_reply":"2024-04-10T13:15:26.871802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\n\n\ntest_segments = int(np.floor(data_test.shape[0] / rows))\nX_test = pd.DataFrame(index=range(test_segments), dtype=np.float64)\ny_test = pd.DataFrame(index=range(test_segments), dtype=np.float64, columns=['time_to_failure'])\nfor segment in range(test_segments):\n    seg = data_test.iloc[segment*rows:segment*rows+rows]\n    extract_features_from_segment(segment, seg, X_test)\n    y_test.loc[segment, 'time_to_failure'] = seg['time_to_failure'].values[-1]\nnew_scaler = StandardScaler()\nnew_scaler.fit(X_test)\nX_test_scaled = new_scaler.transform(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:15:43.182168Z","iopub.execute_input":"2024-04-10T13:15:43.183422Z","iopub.status.idle":"2024-04-10T13:15:54.193436Z","shell.execute_reply.started":"2024-04-10T13:15:43.183379Z","shell.execute_reply":"2024-04-10T13:15:54.192506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Training and results","metadata":{}},{"cell_type":"code","source":"model =   LinearRegression()\nmodel.fit(X_train_scaled, y_train.values.flatten())\ny_pred = model.predict(X_train_scaled)\nscore = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint('Linear Regression on the training data: ' + str(score))","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:23:09.445882Z","iopub.execute_input":"2024-04-10T13:23:09.446495Z","iopub.status.idle":"2024-04-10T13:23:09.471101Z","shell.execute_reply.started":"2024-04-10T13:23:09.446458Z","shell.execute_reply":"2024-04-10T13:23:09.469203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model =   LinearRegression()\nmodel.fit(X_train_scaled, y_train.values.flatten())\ny_pred = model.predict(X_test_scaled)\nscore = mean_absolute_error(y_test.values.flatten(), y_pred)\nprint('Linear Regression on the test data: ' + str(score))","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:23:11.586658Z","iopub.execute_input":"2024-04-10T13:23:11.587126Z","iopub.status.idle":"2024-04-10T13:23:11.601715Z","shell.execute_reply.started":"2024-04-10T13:23:11.587091Z","shell.execute_reply":"2024-04-10T13:23:11.599865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model =   RandomForestRegressor(n_estimators = 100, random_state = 42)\nmodel.fit(X_train_scaled, y_train.values.flatten())\ny_pred = model.predict(X_train_scaled)\nscore = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint('Random Forest Regression on the training data: ' + str(score))","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:23:13.238085Z","iopub.execute_input":"2024-04-10T13:23:13.239629Z","iopub.status.idle":"2024-04-10T13:23:14.240061Z","shell.execute_reply.started":"2024-04-10T13:23:13.239573Z","shell.execute_reply":"2024-04-10T13:23:14.239129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model =   RandomForestRegressor(n_estimators = 100, random_state = 42)\nmodel.fit(X_train_scaled, y_train.values.flatten())\ny_pred = model.predict(X_test_scaled)\nscore = mean_absolute_error(y_test.values.flatten(), y_pred)\nprint('Random Forest Regression on the test data: ' + str(score))","metadata":{"execution":{"iopub.status.busy":"2024-04-10T13:23:16.915817Z","iopub.execute_input":"2024-04-10T13:23:16.916669Z","iopub.status.idle":"2024-04-10T13:23:17.977504Z","shell.execute_reply.started":"2024-04-10T13:23:16.916626Z","shell.execute_reply":"2024-04-10T13:23:17.976346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Discussion and Conclusion\n\nAfter trying out a different features and models, the results presented above are the best that could be achieved. Out of these two, the random foreset regressor is better for both the training and test data. Moreover, a big difference can be seen between the performances of the training data and the test data, which is logical because the model was trained on the training data, but it might be the case that the model is a bit overfitted. If we compare our results to the average results, it is slightly higher but still a pretty good aproximation. To help improve the model, the test data could also be used for training and then the model could be tested on the other csv file. Other features could also be used to improve performance. ","metadata":{}},{"cell_type":"code","source":"\nsubmission = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/sample_submission.csv', index_col='seg_id')\nX_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)\nfor seg_id in X_test.index:\n    seg = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/test/' + seg_id + '.csv')\n    extract_features_from_segment(seg_id, seg, X_test)\n    \nX_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = model.predict(X_test_scaled)\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-10T14:06:33.654339Z","iopub.execute_input":"2024-04-10T14:06:33.654825Z","iopub.status.idle":"2024-04-10T14:08:10.250085Z","shell.execute_reply.started":"2024-04-10T14:06:33.654792Z","shell.execute_reply":"2024-04-10T14:08:10.248014Z"},"trusted":true},"execution_count":null,"outputs":[]}]}