{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Importing Libraries "},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport datetime","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reading train.csv and Train & Test folder"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/train.csv')\n\ntrain_dir = \"../input/predict-volcanic-eruptions-ingv-oe/train/\"\ntest_dir =  \"../input/predict-volcanic-eruptions-ingv-oe/test/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\"Converting time_to_eruption to hours, minutes & seconds\"\"\"\n\ntrain['h:m:s'] = (train['time_to_eruption']\n                  .apply(lambda x:datetime.timedelta(seconds = x/100)))\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> [](http://)**Sample train dataset**"},{"metadata":{"trusted":true},"cell_type":"code","source":"sample = pd.read_csv('../input/predict-volcanic-eruptions-ingv-oe/train/1000015382.csv')\nsample.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Plotting sample data"},{"metadata":{"trusted":true},"cell_type":"code","source":"sample.fillna(0).plot(subplots = True, figsize = (20,15))\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":" **Data for Training Set**"},{"metadata":{"trusted":true},"cell_type":"code","source":"'''Function to get training data'''\ndef get_csv(index):\n    \n    train_data = pd.read_csv(train_dir + str(train.segment_id.iloc[index]) + \".csv\")\n    train_data['time_to_eruption'] = train.time_to_eruption.iloc[index]\n    \n    for feat in train_data.drop('time_to_eruption',1).columns:\n        train_data[feat] = train_data[feat].mean()\n    \n    train_data = train_data.sample()\n    \n    return(train_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.DataFrame()\n\nfor index in range(train.shape[0]):\n    data = pd.concat([get_csv(index), data])\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in data:\n    data[i] = data[i].replace(np.nan, data[i].mean())\n\ndata.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Data for Test Set**"},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv(\"../input/predict-volcanic-eruptions-ingv-oe/sample_submission.csv\")\n\n'''Function to get test data'''\ndef get_csv_test(index):\n    \n    test_data = pd.read_csv(test_dir + str(test.segment_id.iloc[index]) + \".csv\")\n    \n    for feat in test_data.columns:\n        test_data[feat] = test_data[feat].mean()\n    \n    test_data = test_data.sample()\n    \n    return(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test = pd.DataFrame()\n\nfor index in range(test.shape[0]):\n    data_test = pd.concat([get_csv_test(index), data_test])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in data_test:\n    data_test[i] = data_test[i].replace(np.nan, data_test[i].mean())\ndata_test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train = data.drop('time_to_eruption', axis = 1)\ny_train = data.time_to_eruption\nx_test = data_test.copy()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Dimensionality Reduction"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA\nlda = LDA(n_components = 5)\nx_train = lda.fit_transform(x_train, y_train) \nx_test = lda.transform(x_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **XGBoost**"},{"metadata":{"trusted":true},"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBRegressor\n\nmodel = XGBRegressor(max_depth = 10, n_estimators = 20, learning_rate = 0.3)\nmodel.fit(x_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Importance Graph"},{"metadata":{"trusted":true},"cell_type":"code","source":"xgb.plot_importance(model)\nplt.rcParams['figure.figsize'] = [5, 5]\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Predictions"},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = model.predict(x_test)\npred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['time_to_eruption'] = pred\nsub = test[['segment_id', 'time_to_eruption']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}