{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-31T23:11:28.941155Z","iopub.execute_input":"2023-07-31T23:11:28.941578Z","iopub.status.idle":"2023-07-31T23:11:31.985376Z","shell.execute_reply.started":"2023-07-31T23:11:28.941543Z","shell.execute_reply":"2023-07-31T23:11:31.984338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR\nfrom sklearn.metrics import mean_absolute_error","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:11:43.602144Z","iopub.execute_input":"2023-07-31T23:11:43.602533Z","iopub.status.idle":"2023-07-31T23:11:44.248877Z","shell.execute_reply.started":"2023-07-31T23:11:43.602501Z","shell.execute_reply":"2023-07-31T23:11:44.247933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:11:47.666169Z","iopub.execute_input":"2023-07-31T23:11:47.666630Z","iopub.status.idle":"2023-07-31T23:16:09.816647Z","shell.execute_reply.started":"2023-07-31T23:11:47.666591Z","shell.execute_reply":"2023-07-31T23:16:09.815410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a training file with simple derived features\n\nrows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['ave', 'std', 'max', 'min'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    \n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:16:14.680859Z","iopub.execute_input":"2023-07-31T23:16:14.681276Z","iopub.status.idle":"2023-07-31T23:16:20.727911Z","shell.execute_reply.started":"2023-07-31T23:16:14.681241Z","shell.execute_reply":"2023-07-31T23:16:20.726854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\nsvm = NuSVR()\nsvm.fit(X_train_scaled, y_train.values.flatten())\ny_pred = svm.predict(X_train_scaled)\nplt.figure(figsize=(6, 6))\nplt.scatter(y_train.values.flatten(), y_pred)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.ylabel('predicted', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\nplt.show()\nscore = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint(f'Score: {score:0.3f}')\nsubmission = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/sample_submission.csv', index_col='seg_id')\nX_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)\nfor seg_id in X_test.index:\n    seg = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()\nX_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = svm.predict(X_test_scaled)\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:18:19.824401Z","iopub.execute_input":"2023-07-31T23:18:19.824792Z","iopub.status.idle":"2023-07-31T23:19:37.939420Z","shell.execute_reply.started":"2023-07-31T23:18:19.824761Z","shell.execute_reply":"2023-07-31T23:19:37.938391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:19:52.127530Z","iopub.execute_input":"2023-07-31T23:19:52.127945Z","iopub.status.idle":"2023-07-31T23:20:01.291504Z","shell.execute_reply.started":"2023-07-31T23:19:52.127908Z","shell.execute_reply":"2023-07-31T23:20:01.290392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras as keras\nimport numpy as np\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.optimizers import SGD, Adam","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:20:10.251439Z","iopub.execute_input":"2023-07-31T23:20:10.251838Z","iopub.status.idle":"2023-07-31T23:20:10.256957Z","shell.execute_reply.started":"2023-07-31T23:20:10.251807Z","shell.execute_reply":"2023-07-31T23:20:10.255951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"layers = [\n    keras.layers.Dense(100, activation = 'relu', input_shape= (4,)),\n    keras.layers.Dropout(0.1),\n    keras.layers.Dense(100, activation = 'tanh'),\n    keras.layers.Dense(100, activation = 'relu'),\n    keras.layers.Dense(1, activation = 'linear'),\n]\n\nmodel = keras.Sequential(layers)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-30T21:55:43.216940Z","iopub.status.idle":"2023-07-30T21:55:43.217646Z","shell.execute_reply.started":"2023-07-30T21:55:43.217395Z","shell.execute_reply":"2023-07-30T21:55:43.217418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Newer version of the above, with new LSTM layers\ncombinations_to_try=[]\nfor layerLS1 in [128,256,512]:\n    for layerLS2 in [128,256,512]:\n        for function1 in ['relu','sigmoid','tanh']:\n            for function2 in ['relu','sigmoid','tanh']:\n                combinations_to_try.append([layerLS1,layerLS2,function1,function2])\nresults=[]\nfor params in combinations_to_try:\n    newLayers = [\n        keras.layers.LSTM(params[0], activation = params[2], input_shape= (4,1), return_sequences=True),\n        keras.layers.LSTM(params[1], activation = params[3], return_sequences=True),\n        keras.layers.Dense(100, activation = 'relu'),\n        keras.layers.Dropout(0.1),\n        keras.layers.Dense(100, activation = 'tanh'),\n        keras.layers.Dense(100, activation = 'relu'),\n        keras.layers.Dense(1, activation = 'linear'),\n    ]\n\n    modelNew = keras.Sequential(newLayers)\n    #modelNew.summary()\n    learning_rate = 1e-3\n    optimizer = SGD(learning_rate = learning_rate)\n    modelNew.compile(optimizer=optimizer, loss = 'mae', metrics=['mae','mse'])\n    n_epochs = 400\n    history = modelNew.fit(X_train_scaled, y_train.values.flatten(), batch_size=1000, epochs=n_epochs, verbose=0)\n    print(params)\n    print(f\"Training loss on the final epoch was: {history.history['loss'][-1]:0.4f}\")\n    results.append(modelNew.evaluate(X_train_scaled, y_train.values.flatten()))\nfor index in range(len(results)):\n    print(results[index])\n    print(combinations_to_try[index])","metadata":{"execution":{"iopub.status.busy":"2023-07-30T22:38:29.153307Z","iopub.execute_input":"2023-07-30T22:38:29.153705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterations related to the Dense layers (LSTM assumed to have failed)\ncombinations_to_try=[]\nfor layerLS1 in [128,256]:\n    for layerLS2 in [128,256]:\n        for layerLS3 in [128,256]:\n            for function1 in ['relu','sigmoid','tanh']:\n                for function2 in ['relu','sigmoid','tanh']:\n                    for function3 in ['relu','sigmoid','tanh']:\n                        combinations_to_try.append([layerLS1,layerLS2,layerLS3,function1,function2,function3])\nresults=[]\nfor params in combinations_to_try:\n    newLayers = [\n        keras.layers.Dense(params[0], activation = params[3], input_shape=(4,)),\n        keras.layers.Dropout(0.1),\n        keras.layers.Dense(params[1], activation = params[4]),\n        keras.layers.Dense(params[2], activation = params[5]),\n        keras.layers.Dense(1, activation = 'linear')\n    ]\n\n    modelNew = keras.Sequential(newLayers)\n    #modelNew.summary()\n    learning_rate = 1e-3\n    optimizer = SGD(learning_rate = learning_rate)\n    modelNew.compile(optimizer=optimizer, loss = 'mae', metrics=['mae','mse'])\n    n_epochs = 400\n    history = modelNew.fit(X_train_scaled, y_train.values.flatten(), batch_size=1000, epochs=n_epochs, verbose=0)\n    print(params)\n    print(f\"Training loss on the final epoch was: {history.history['loss'][-1]:0.4f}\")\n    results.append(modelNew.evaluate(X_train_scaled, y_train.values.flatten()))\nfor index in range(len(results)):\n    print(results[index])\n    print(combinations_to_try[index])","metadata":{"execution":{"iopub.status.busy":"2023-07-31T23:31:47.166346Z","iopub.execute_input":"2023-07-31T23:31:47.166736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Try: Learning Rate= 0.1,0.05,0.02,0.01,0.005,0.002,0.001,0.0005,0.0002,0.0001 (DEFAULT 1e-2)\nlearning_rate = 1e-3\noptimizer = SGD(learning_rate = learning_rate)\nmodel.compile(optimizer=optimizer, loss = 'mae', metrics=['mae','mse'])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-30T20:39:19.421023Z","iopub.execute_input":"2023-07-30T20:39:19.421434Z","iopub.status.idle":"2023-07-30T20:39:19.444268Z","shell.execute_reply.started":"2023-07-30T20:39:19.421401Z","shell.execute_reply":"2023-07-30T20:39:19.441563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(results)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T01:18:29.063400Z","iopub.execute_input":"2023-07-31T01:18:29.064249Z","iopub.status.idle":"2023-07-31T01:18:29.542892Z","shell.execute_reply.started":"2023-07-31T01:18:29.064200Z","shell.execute_reply":"2023-07-31T01:18:29.541470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_epochs = 400\nhistory = model.fit(X_train_scaled, y_train.values.flatten(), batch_size=1000, epochs=n_epochs)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T20:39:19.446830Z","iopub.execute_input":"2023-07-30T20:39:19.447235Z","iopub.status.idle":"2023-07-30T20:39:40.689622Z","shell.execute_reply.started":"2023-07-30T20:39:19.447202Z","shell.execute_reply":"2023-07-30T20:39:40.687941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(np.arange(1, n_epochs+1), history.history['loss'])\nplt.title('Training set loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\n\nplt.subplot(1, 2, 2)\nplt.semilogy(np.arange(1, n_epochs+1), history.history['mae'], label='mae')\nplt.semilogy(np.arange(1, n_epochs+1), history.history['mse'], label='mse')\nplt.legend()\nplt.title('Training set metric scores')\nplt.xlabel('Epoch')\nplt.ylabel('Error')\n\nprint(f\"Training loss on the final epoch was: {history.history['loss'][-1]:0.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-30T20:39:40.691837Z","iopub.execute_input":"2023-07-30T20:39:40.692221Z","iopub.status.idle":"2023-07-30T20:39:41.313343Z","shell.execute_reply.started":"2023-07-30T20:39:40.692189Z","shell.execute_reply":"2023-07-30T20:39:41.311434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(X_train_scaled, y_train.values.flatten())","metadata":{"execution":{"iopub.status.busy":"2023-07-30T20:39:41.315673Z","iopub.execute_input":"2023-07-30T20:39:41.316015Z","iopub.status.idle":"2023-07-30T20:39:41.766505Z","shell.execute_reply.started":"2023-07-30T20:39:41.315989Z","shell.execute_reply":"2023-07-30T20:39:41.765151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_train_scaled)\nplt.figure(figsize=(6, 6))\nplt.scatter(y_train.values.flatten(), y_pred)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.ylabel('predicted', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\nplt.show()\nscore = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint(f'Score: {score:0.3f}')\nsubmission = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/sample_submission.csv', index_col='seg_id')\nX_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)\nfor seg_id in X_test.index:\n    seg = pd.read_csv('/kaggle/input/LANL-Earthquake-Prediction/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()\nX_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = svm.predict(X_test_scaled)\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-30T20:39:41.767731Z","iopub.execute_input":"2023-07-30T20:39:41.768022Z","iopub.status.idle":"2023-07-30T20:40:17.837498Z","shell.execute_reply.started":"2023-07-30T20:39:41.767996Z","shell.execute_reply":"2023-07-30T20:40:17.836164Z"},"trusted":true},"execution_count":null,"outputs":[]}]}