{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This is second version of this project. My main purposes are to try many different ML algorithms, approaches, and ways of dealing with problems. I am begginer in ML, so if you have cought any errors or have a suggestion, feel free to write it in a comment :)"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_dir = '/kaggle/input/predict-volcanic-eruptions-ingv-oe/train/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# function to read data\n\ndef combine_csv():\n    def get_data(index):\n        data = pd.read_csv(data_dir + str(train.segment_id.iloc[index]) + \".csv\")\n        data['time_to_eruption'] = train['time_to_eruption'].iloc[index]\n        for feat in data.drop('time_to_eruption',1).columns:\n            data[feat] = data[feat].mean()\n        data = data.sample(1)\n        return data\n    \n    def combine():\n        data = pd.DataFrame()\n        for i in range(train.shape[0]):\n            df = get_data(i)\n            data=pd.concat([df,data])\n        return data\n    \n    return combine()\n\ndata = combine_csv()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# filling null vals with median\n\nfor col in data.columns:\n    median = data[col].median()\n    data[col].fillna(median, inplace=True)\n    \ndata.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_mx = data.corr()\nsns.heatmap(corr_mx)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Modeling\n\nfrom sklearn.model_selection import train_test_split\n\nX = data.drop(['time_to_eruption'], inplace=False, axis=1)\ny = data['time_to_eruption']\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\nX_train.shape\nX_train = np.array(X_train)\ny_train = np.array(y_train)\n\nX_train=X_train.reshape(-1,10,1)\n\nX_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.svm import SVR\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.decomposition import PCA, KernelPCA\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.cluster import KMeans, DBSCAN, MiniBatchKMeans\nfrom sklearn.ensemble import AdaBoostRegressor, VotingRegressor, BaggingRegressor, RandomForestRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow import keras\nfrom tensorflow.keras.optimizers import SGD\n\n#model = keras.models.Sequential()\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(2835, activation = \"relu\"))\n#model.add(keras.layers.Dense(1, activation = \"linear\"))\n\n#sgd = SGD(learning_rate = 0.001)\n#model.compile(optimizer= 'adam' ,loss= 'mean_absolute_error',metrics=['mean_absolute_error'])\n#model.fit(X_train, y_train,epochs=5, batch_size=250, validation_data=(X_val, y_val))\n\n\nmodel = keras.models.Sequential([\nkeras.layers.Conv1D(128, 2, activation = \"relu\", padding = 'valid'),\nkeras.layers.MaxPool1D(2),\nkeras.layers.Conv1D(128, 2, activation = \"relu\", padding = 'valid'),\nkeras.layers.Conv1D(128, 2, activation = \"relu\", padding = 'valid'),\nkeras.layers.MaxPool1D(2),\nkeras.layers.Flatten(),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(128, activation = \"relu\"),\nkeras.layers.Dense(64, activation = \"relu\"),\nkeras.layers.Dense(1)\n])\nsgd = SGD(learning_rate = 0.001)\nmodel.compile(optimizer= 'adam' ,loss= 'mean_absolute_error',metrics=['mean_absolute_error'])\nmodel.fit(X_train, y_train, epochs=30, batch_size=5)\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"First thing i want to see is how this data reacts to reductions in dimensionality, since this is a topic i've recently learned, and i can't wait to see what i can do with it! ( I still don't understand it fully, so for this version my explorations are left quite shallow )"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Not much difrence compared to PCA. Later in this notebook I'll see how PCA/KPCA scales with my other methods."},{"metadata":{},"cell_type":"markdown","source":"\nAlso recently, I've read about data preprocessing with clustering algorithms, and I am determined to find something usefull here!"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Better that just PCA, but worse that just KMeans, which means that dimensionality reduction in case of this model in counterproductive"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# print(f'Decision Tree: {kmeans(dtr)}\\nSupport Vector Machines: {kmeans(SVR())}\\nLinear Regression: {kmeans(LinearRegression())}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Best params:\n# Decision Tree: (173512024577686.5, {'kmeans__n_clusters': 450})\n# Support Vector Machines: (182206672899156.47, {'kmeans__n_clusters': 1500})\n# Linear Regression: (6.241083175756122e+16, {'kmeans__n_clusters': 150})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# I noticed that mse seems to be minimized at around n_clusters=1000 ( Lets forget linear regression )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Best params:\n# Decision Tree: (170436558910960.75, {'minikmeans__batch_size': 500, 'minikmeans__n_clusters': 100})\n# Support Vector Machines: (182206653608092.66, {'minikmeans__batch_size': 700, 'minikmeans__n_clusters': 1})\n# Linear Regression: (164458165654294.5, {'minikmeans__batch_size': 200, 'minikmeans__n_clusters': 100})\n# Linear regression seems to improve the most, later might tweak n_clusters_param","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems that MiniBatchKMeans works better that regular KMeans"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Ensemble learning**"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv\")\n\n'''Function to get test data'''\ndef get_csv_test(index):\n    \n    test_data = pd.read_csv(test_dir + str(test.segment_id.iloc[index]) + \".csv\")\n    \n    for feat in test_data.columns:\n        test_data[feat] = test_data[feat].mean()\n    \n    test_data = test_data.sample()\n    \n    return(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test = pd.DataFrame()\ntest_dir = '/kaggle/input/predict-volcanic-eruptions-ingv-oe/test/'\n\nfor index in range(test.shape[0]):\n    data_test = pd.concat([get_csv_test(index), data_test])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test.shape\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in data_test:\n    data_test[i] = data_test[i].replace(np.nan, data_test[i].mean())\ndata_test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = ['sensor_1','sensor_2','sensor_3','sensor_4','sensor_5','sensor_6','sensor_7','sensor_8','sensor_9','sensor_10']\ntest_X = data_test[X]\ntest_X = np.array(test_X)\ntest_X=test_X.reshape(-1,10,1)\npredicted_time = model.predict(test_X)\nprint(predicted_time)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['time_to_eruption'] = predicted_time\nsub = test[['segment_id', 'time_to_eruption']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()\nsub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}