{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objs as go","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/train.csv\")\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(\n    train, \n    x=\"time_to_eruption\",\n    width=800,\n    height=500,\n    nbins=100,\n    title='Time to eruption distribution'\n)\n\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(\n    train, \n    y=\"time_to_eruption\",\n    width=800,\n    height=500,\n    title='Time to eruption for all volcanos'\n)\n\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['time_to_eruption'].describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"../input/predict-volcanic-eruptions-ingv-oe/train/\"\ntest_dir = \"../input/predict-volcanic-eruptions-ingv-oe/test/\" ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_csv(index):\n    train1 = pd.read_csv(train_dir + str(train.segment_id.iloc[index]) + \".csv\")\n\n    train1['timetoerupt'] = train.time_to_eruption.iloc[index]\n    \n    for feat in train1.drop('timetoerupt',1).columns:\n        train1[feat] = train1[feat].mean()\n    \n    train1 = train1.sample(1)\n           \n    return (train1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame()\n\nfor idx in range(train.shape[0]):\n   df = read_csv(idx)\n    \n   data=pd.concat([df,data])\n\ndata.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feat in data:\n    data[feat] = data[feat].replace(np.nan, data[feat].mean())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL BUILDING","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model\nfrom sklearn.linear_model import LinearRegression\n\nimport statsmodels.api as sm\n\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(data, data.timetoerupt, test_size=0.2, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.drop('timetoerupt',1,inplace = True)\n\n# Add a constant to get an intercept\nX_train_sm = sm.add_constant(X_train)\n\n# train the model\nlr = sm.OLS(y_train, X_train_sm).fit()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.drop('timetoerupt',1,inplace = True)\n\n# Add a constant to get an intercept\nX_test_sm = sm.add_constant(X_test)\n\n# prediction on training dataset\ny_test_pred = lr.predict(X_test_sm)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(\n    y_test_pred, \n    x=\"time_to_eruption\",\n    width=800,\n    height=500,\n    nbins=100,\n    title='Time to eruption distribution'\n)\n\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nimport math","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mae = np.mean(np.abs(y_test - y_test_pred))\nprint(\"Mean Absolute Error:\", mae)\n\nR2 = np.corrcoef(y_test,y_test_pred)[0,1]**2\nprint(\"R2 Score :\", R2)\n\nMSE = np.square(np.subtract(y_test,y_test_pred)).mean() \nRMSE = math.sqrt(MSE)\nprint(\"Root Mean Squared Error :\", RMSE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}