{"cells":[{"metadata":{"papermill":{"duration":0.026027,"end_time":"2020-10-17T14:39:55.246921","exception":false,"start_time":"2020-10-17T14:39:55.220894","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Approach:\n\n1) Read a csv file given in \"train\" folder, take mean of all features and put it into a dataframe.\n2) Add \"time to erupt\" feature from the train.csv file to the dataframe created in step # 1\n3) repeat steps # 1 and 2 for all the csv files present in \"train\" folder\n\nAt the end of step3, we will have a dataframe which would contain data for all the segments, mean of recordings from all the censors. I saved this file and loaded back into \"../input/volcano-eruption-data\" and thats why you may find some code has been commented out to save execution time.\n\nNext, I have built a naive model to predict the \"time to erupt\" for test data."},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:55.304672Z","iopub.status.busy":"2020-10-17T14:39:55.303889Z","iopub.status.idle":"2020-10-17T14:39:56.407352Z","shell.execute_reply":"2020-10-17T14:39:56.406657Z"},"papermill":{"duration":1.133646,"end_time":"2020-10-17T14:39:56.407514","exception":false,"start_time":"2020-10-17T14:39:55.273868","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-10-17T14:39:56.462463Z","iopub.status.busy":"2020-10-17T14:39:56.461420Z","iopub.status.idle":"2020-10-17T14:39:56.464180Z","shell.execute_reply":"2020-10-17T14:39:56.464772Z"},"papermill":{"duration":0.032253,"end_time":"2020-10-17T14:39:56.464938","exception":false,"start_time":"2020-10-17T14:39:56.432685","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#train = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/train.csv\")\n#train.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:56.520194Z","iopub.status.busy":"2020-10-17T14:39:56.519424Z","iopub.status.idle":"2020-10-17T14:39:56.522677Z","shell.execute_reply":"2020-10-17T14:39:56.521986Z"},"papermill":{"duration":0.032925,"end_time":"2020-10-17T14:39:56.522808","exception":false,"start_time":"2020-10-17T14:39:56.489883","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train_dir = \"../input/predict-volcanic-eruptions-ingv-oe/train/\"\ntest_dir = \"../input/predict-volcanic-eruptions-ingv-oe/test/\"    ","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.02473,"end_time":"2020-10-17T14:39:56.572695","exception":false,"start_time":"2020-10-17T14:39:56.547965","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### Helper function to read csv files present in the \"train\" folder"},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:56.631778Z","iopub.status.busy":"2020-10-17T14:39:56.630702Z","iopub.status.idle":"2020-10-17T14:39:56.633850Z","shell.execute_reply":"2020-10-17T14:39:56.634410Z"},"papermill":{"duration":0.036831,"end_time":"2020-10-17T14:39:56.634603","exception":false,"start_time":"2020-10-17T14:39:56.597772","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"def read_csv(index):\n    train1 = pd.read_csv(train_dir + str(train.segment_id.iloc[index]) + \".csv\")\n\n    train1['timetoerupt'] = train.time_to_eruption.iloc[index]\n    \n    for feat in train1.drop('timetoerupt',1).columns:\n        train1[feat] = train1[feat].mean()\n    \n    train1 = train1.sample(1)\n           \n    return (train1)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.025032,"end_time":"2020-10-17T14:39:56.684838","exception":false,"start_time":"2020-10-17T14:39:56.659806","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### Read the files and create a dataframe"},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:56.740188Z","iopub.status.busy":"2020-10-17T14:39:56.739133Z","iopub.status.idle":"2020-10-17T14:39:56.742729Z","shell.execute_reply":"2020-10-17T14:39:56.742113Z"},"papermill":{"duration":0.033014,"end_time":"2020-10-17T14:39:56.742856","exception":false,"start_time":"2020-10-17T14:39:56.709842","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#data = pd.DataFrame()\n\n#for idx in range(train.shape[0]):\n#    df = read_csv(idx)\n    \n#    data=pd.concat([df,data])","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.028192,"end_time":"2020-10-17T14:39:56.797138","exception":false,"start_time":"2020-10-17T14:39:56.768946","status":"completed"},"tags":[]},"cell_type":"markdown","source":"I have already ran the above steps, created the dataframe, saved it into a csv file, and loaded it back for further use.\nWe will load the same file below."},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:56.854860Z","iopub.status.busy":"2020-10-17T14:39:56.853739Z","iopub.status.idle":"2020-10-17T14:39:56.904714Z","shell.execute_reply":"2020-10-17T14:39:56.903994Z"},"papermill":{"duration":0.082125,"end_time":"2020-10-17T14:39:56.904835","exception":false,"start_time":"2020-10-17T14:39:56.822710","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# load training data\ndata = pd.read_csv(\"../input/volcano-eruption-data/data.csv\")\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:56.964646Z","iopub.status.busy":"2020-10-17T14:39:56.963812Z","iopub.status.idle":"2020-10-17T14:39:56.967865Z","shell.execute_reply":"2020-10-17T14:39:56.967226Z"},"papermill":{"duration":0.035533,"end_time":"2020-10-17T14:39:56.967986","exception":false,"start_time":"2020-10-17T14:39:56.932453","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# this will confirm whether we have read all the files or not\ndata.shape","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:57.030465Z","iopub.status.busy":"2020-10-17T14:39:57.029373Z","iopub.status.idle":"2020-10-17T14:39:57.036609Z","shell.execute_reply":"2020-10-17T14:39:57.037160Z"},"papermill":{"duration":0.041147,"end_time":"2020-10-17T14:39:57.037353","exception":false,"start_time":"2020-10-17T14:39:56.996206","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:57.100853Z","iopub.status.busy":"2020-10-17T14:39:57.099671Z","iopub.status.idle":"2020-10-17T14:39:57.110101Z","shell.execute_reply":"2020-10-17T14:39:57.110655Z"},"papermill":{"duration":0.045956,"end_time":"2020-10-17T14:39:57.110823","exception":false,"start_time":"2020-10-17T14:39:57.064867","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# replace null values with the mean value\nfor feat in data:\n    data[feat] = data[feat].replace(np.nan, data[feat].mean())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:57.172147Z","iopub.status.busy":"2020-10-17T14:39:57.171356Z","iopub.status.idle":"2020-10-17T14:39:57.179148Z","shell.execute_reply":"2020-10-17T14:39:57.178205Z"},"papermill":{"duration":0.040792,"end_time":"2020-10-17T14:39:57.179326","exception":false,"start_time":"2020-10-17T14:39:57.138534","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.027796,"end_time":"2020-10-17T14:39:57.235829","exception":false,"start_time":"2020-10-17T14:39:57.208033","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Model Building"},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:57.298714Z","iopub.status.busy":"2020-10-17T14:39:57.297991Z","iopub.status.idle":"2020-10-17T14:39:58.313080Z","shell.execute_reply":"2020-10-17T14:39:58.312341Z"},"papermill":{"duration":1.048445,"end_time":"2020-10-17T14:39:58.313215","exception":false,"start_time":"2020-10-17T14:39:57.264770","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"from sklearn import linear_model\nfrom sklearn.linear_model import LinearRegression\n\nimport statsmodels.api as sm\n\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.380798Z","iopub.status.busy":"2020-10-17T14:39:58.376649Z","iopub.status.idle":"2020-10-17T14:39:58.384200Z","shell.execute_reply":"2020-10-17T14:39:58.383384Z"},"papermill":{"duration":0.042514,"end_time":"2020-10-17T14:39:58.384320","exception":false,"start_time":"2020-10-17T14:39:58.341806","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(data, data.timetoerupt, test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.449665Z","iopub.status.busy":"2020-10-17T14:39:58.448821Z","iopub.status.idle":"2020-10-17T14:39:58.548033Z","shell.execute_reply":"2020-10-17T14:39:58.547060Z"},"papermill":{"duration":0.134908,"end_time":"2020-10-17T14:39:58.548200","exception":false,"start_time":"2020-10-17T14:39:58.413292","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"X_train.drop('timetoerupt',1,inplace = True)\n\n# Add a constant to get an intercept\nX_train_sm = sm.add_constant(X_train)\n\n# train the model\nlr = sm.OLS(y_train, X_train_sm).fit()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.613079Z","iopub.status.busy":"2020-10-17T14:39:58.612222Z","iopub.status.idle":"2020-10-17T14:39:58.631702Z","shell.execute_reply":"2020-10-17T14:39:58.632570Z"},"papermill":{"duration":0.054473,"end_time":"2020-10-17T14:39:58.632757","exception":false,"start_time":"2020-10-17T14:39:58.578284","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"print(lr.summary())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.707123Z","iopub.status.busy":"2020-10-17T14:39:58.706204Z","iopub.status.idle":"2020-10-17T14:39:58.710246Z","shell.execute_reply":"2020-10-17T14:39:58.709626Z"},"papermill":{"duration":0.045078,"end_time":"2020-10-17T14:39:58.710372","exception":false,"start_time":"2020-10-17T14:39:58.665294","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"X_test.drop('timetoerupt',1,inplace = True)\n\n# Add a constant to get an intercept\nX_test_sm = sm.add_constant(X_test)\n\n# prediction on training dataset\ny_test_pred = lr.predict(X_test_sm)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.784344Z","iopub.status.busy":"2020-10-17T14:39:58.783135Z","iopub.status.idle":"2020-10-17T14:39:58.789602Z","shell.execute_reply":"2020-10-17T14:39:58.790407Z"},"papermill":{"duration":0.050191,"end_time":"2020-10-17T14:39:58.790652","exception":false,"start_time":"2020-10-17T14:39:58.740461","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.880971Z","iopub.status.busy":"2020-10-17T14:39:58.879796Z","iopub.status.idle":"2020-10-17T14:39:58.886147Z","shell.execute_reply":"2020-10-17T14:39:58.885299Z"},"papermill":{"duration":0.056137,"end_time":"2020-10-17T14:39:58.886294","exception":false,"start_time":"2020-10-17T14:39:58.830157","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"r_squared = r2_score(y_test_pred, y_test)\nr_squared","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:58.959651Z","iopub.status.busy":"2020-10-17T14:39:58.958850Z","iopub.status.idle":"2020-10-17T14:39:58.970903Z","shell.execute_reply":"2020-10-17T14:39:58.970232Z"},"papermill":{"duration":0.051372,"end_time":"2020-10-17T14:39:58.971034","exception":false,"start_time":"2020-10-17T14:39:58.919662","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sub = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv\")\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.031152,"end_time":"2020-10-17T14:39:59.033980","exception":false,"start_time":"2020-10-17T14:39:59.002828","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### Helper function to read csv files in the test folder"},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.105333Z","iopub.status.busy":"2020-10-17T14:39:59.104601Z","iopub.status.idle":"2020-10-17T14:39:59.107649Z","shell.execute_reply":"2020-10-17T14:39:59.106877Z"},"papermill":{"duration":0.042367,"end_time":"2020-10-17T14:39:59.107779","exception":false,"start_time":"2020-10-17T14:39:59.065412","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"def read_csv(index):\n    \n    test1 = pd.read_csv(test_dir + str(sub.segment_id.iloc[index]) + \".csv\")\n\n    for feat in test1.columns:\n        test1[feat] = test1[feat].mean()\n    \n    test1 = test1.sample(1)\n           \n    return (test1)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.178071Z","iopub.status.busy":"2020-10-17T14:39:59.177058Z","iopub.status.idle":"2020-10-17T14:39:59.180956Z","shell.execute_reply":"2020-10-17T14:39:59.180263Z"},"papermill":{"duration":0.041017,"end_time":"2020-10-17T14:39:59.181091","exception":false,"start_time":"2020-10-17T14:39:59.140074","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#test = pd.DataFrame()\n\n#for idx in range(sub.shape[0]):\n#    df = read_csv(idx)\n    \n#    test = pd.concat([df,test])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.253131Z","iopub.status.busy":"2020-10-17T14:39:59.252355Z","iopub.status.idle":"2020-10-17T14:39:59.288700Z","shell.execute_reply":"2020-10-17T14:39:59.287876Z"},"papermill":{"duration":0.07446,"end_time":"2020-10-17T14:39:59.288835","exception":false,"start_time":"2020-10-17T14:39:59.214375","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# I have ran the steps mentioned above and saved the file, loading it now\ntest = pd.read_csv(\"../input/volcano-eruption-data/test.csv\")\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.360248Z","iopub.status.busy":"2020-10-17T14:39:59.359428Z","iopub.status.idle":"2020-10-17T14:39:59.363038Z","shell.execute_reply":"2020-10-17T14:39:59.363650Z"},"papermill":{"duration":0.042067,"end_time":"2020-10-17T14:39:59.363828","exception":false,"start_time":"2020-10-17T14:39:59.321761","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# again verify whether all the files were read correctly or not\ntest.shape","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.438754Z","iopub.status.busy":"2020-10-17T14:39:59.436564Z","iopub.status.idle":"2020-10-17T14:39:59.442770Z","shell.execute_reply":"2020-10-17T14:39:59.443299Z"},"papermill":{"duration":0.046044,"end_time":"2020-10-17T14:39:59.443463","exception":false,"start_time":"2020-10-17T14:39:59.397419","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.523996Z","iopub.status.busy":"2020-10-17T14:39:59.522713Z","iopub.status.idle":"2020-10-17T14:39:59.529176Z","shell.execute_reply":"2020-10-17T14:39:59.528506Z"},"papermill":{"duration":0.051697,"end_time":"2020-10-17T14:39:59.529307","exception":false,"start_time":"2020-10-17T14:39:59.477610","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# same as we did for the training data\nfor feat in test:\n    test[feat] = test[feat].replace(np.nan, test[feat].mean())","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.603336Z","iopub.status.busy":"2020-10-17T14:39:59.602607Z","iopub.status.idle":"2020-10-17T14:39:59.605667Z","shell.execute_reply":"2020-10-17T14:39:59.604924Z"},"papermill":{"duration":0.04218,"end_time":"2020-10-17T14:39:59.605794","exception":false,"start_time":"2020-10-17T14:39:59.563614","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"#test.to_csv('test.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.683685Z","iopub.status.busy":"2020-10-17T14:39:59.680639Z","iopub.status.idle":"2020-10-17T14:39:59.690056Z","shell.execute_reply":"2020-10-17T14:39:59.689196Z"},"papermill":{"duration":0.050067,"end_time":"2020-10-17T14:39:59.690201","exception":false,"start_time":"2020-10-17T14:39:59.640134","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# Add a constant to get an intercept\ntest_sm = sm.add_constant(test)\n\n# prediction on test dataset\npredictions = lr.predict(test_sm)\n\nsub['time_to_eruption'] = predictions","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.774359Z","iopub.status.busy":"2020-10-17T14:39:59.773408Z","iopub.status.idle":"2020-10-17T14:39:59.778275Z","shell.execute_reply":"2020-10-17T14:39:59.777523Z"},"papermill":{"duration":0.050722,"end_time":"2020-10-17T14:39:59.778424","exception":false,"start_time":"2020-10-17T14:39:59.727702","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-10-17T14:39:59.859096Z","iopub.status.busy":"2020-10-17T14:39:59.858261Z","iopub.status.idle":"2020-10-17T14:40:00.187432Z","shell.execute_reply":"2020-10-17T14:40:00.186616Z"},"papermill":{"duration":0.369821,"end_time":"2020-10-17T14:40:00.187600","exception":false,"start_time":"2020-10-17T14:39:59.817779","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# submission file\nsub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}