{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nimport pandas as pd\n\nfrom time import time\nfrom time import ctime\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom tqdm import tqdm_notebook\nfrom tqdm import tqdm\nimport lightgbm as lgb\nimport joblib\nfrom joblib import Parallel, delayed\nimport multiprocessing\nnum_cores = multiprocessing.cpu_count()-1\n\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import r2_score\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import KFold\nimport matplotlib.pyplot as plt\n\ndef plotfig (ypred, yactual, strtitle, y_max):\n    plt.scatter(ypred, yactual.values.ravel())\n    plt.title(strtitle)\n    plt.plot([(0, 0), (y_max, y_max)], [(0, 0), (y_max, y_max)])\n    plt.xlim(0, y_max)\n    plt.ylim(0, y_max)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('Actual', fontsize=12)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/ingv-tsfresh-7730/train.csv', sep = ';')\ntrain.set_index('Unnamed: 0', inplace = True)\ntest = pd.read_csv('../input/ingv-tsfresh-7730/test.csv', sep = ';')\ntest.set_index('Unnamed: 0', inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_index = test.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_rf = train.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_rf = train_rf.fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_rf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_rf.drop('time_to_eruption', axis=1)\ny = train_rf['time_to_eruption']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(random_state=1, max_depth=7)\nmodel.fit(x,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_scores = pd.Series(model.feature_importances_, index=x.columns).sort_values(ascending=False)\nfeature_scores","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"selected_feature = feature_scores[:350].index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target = train['time_to_eruption']\nall_data = pd.concat([train, test], ignore_index = True)\nall_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_data = pd.concat([all_data[selected_feature], all_data['time_to_eruption']], axis=1)\nall_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Function to calculate missing values by column# Funct \ndef missing_values_table(df):\n        # Total missing values\n        mis_val = df.isnull().sum()\n        \n        # Percentage of missing values\n        mis_val_percent = 100 * df.isnull().sum() / len(df)\n        \n        # Make a table with the results\n        mis_val_table = pd.concat([mis_val, mis_val_percent], axis=1)\n        \n        # Rename the columns\n        mis_val_table_ren_columns = mis_val_table.rename(\n        columns = {0 : 'Missing Values', 1 : '% of Total Values'})\n        \n        # Sort the table by percentage of missing descending\n        mis_val_table_ren_columns = mis_val_table_ren_columns[\n            mis_val_table_ren_columns.iloc[:,1] != 0].sort_values(\n        '% of Total Values', ascending=False).round(1)\n        \n        # Print some summary information\n        print (\"Your selected dataframe has \" + str(df.shape[1]) + \" columns.\\n\"      \n            \"There are \" + str(mis_val_table_ren_columns.shape[0]) +\n              \" columns that have missing values.\")\n        \n        # Return the dataframe with missing information\n        return mis_val_table_ren_columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"missing_values = missing_values_table(all_data)\nmissing_values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_data = all_data.fillna(all_data.mode())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"header = all_data.columns\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nall_data[header] = scaler.fit_transform(all_data)\nall_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_data = all_data.drop('time_to_eruption', axis=1)\nall_data.var()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_data.corr()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"missing_values = missing_values_table(all_data)\nmissing_values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_data = all_data.fillna(all_data.min())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# from sklearn.decomposition import KernelPCA \nfrom sklearn.decomposition import PCA\npca = PCA() \nall_data_pca = pca.fit_transform(all_data)\nall_data_pca = pd.DataFrame(all_data_pca)\nall_data_pca.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = all_data_pca[:train.shape[0]]\ntest = all_data_pca[train.shape[0]:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y = target\nX = train\nX_test = test\n\nn_fold = 5\ncv = KFold(n_splits=n_fold, shuffle=True, random_state=123)\n\noof = np.zeros(len(X))\ncat_prediction = np.zeros(len(X_test))\nmae, r2 = [], []\n\nPARAMS = {\n\n        'random_seed': 123,\n        'eval_metric': 'MAE',\n        'task_type': 'GPU',\n        'bagging_temperature': 0.41010395885331385,\n        'border_count': 186,\n        'depth': 10,\n        'iterations': 700,\n        'l2_leaf_reg': 30,\n        'learning_rate': 0.067,\n        'random_strength': 3.230824361824754e-06,\n    }\n\nfor fold_n, (train_index, valid_index) in enumerate(cv.split(X)):\n    print('\\nFold', fold_n, 'started at', ctime())\n\n    X_train = X.iloc[train_index,:]\n    X_valid = X.iloc[valid_index,:]\n    \n    Y_train = Y.iloc[train_index]\n    Y_valid = Y.iloc[valid_index]\n          \n    best_model = CatBoostRegressor(**PARAMS, thread_count = -1)  \n    \n    train_dataset = Pool(data=X_train,\n                     label=Y_train,\n                     )\n    \n    eval_dataset = Pool(data=X_valid,\n                    label=Y_valid,\n                    )\n    \n    best_model.fit(train_dataset,\n              use_best_model=True,\n              verbose = False,\n              plot = True,\n              eval_set=eval_dataset,\n              early_stopping_rounds=70)\n\n   \n    y_pred = best_model.predict(Pool(data=X_valid))\n\n    mae.append(mean_absolute_error(Y_valid, y_pred))\n    r2.append(r2_score(Y_valid, y_pred))\n\n    print('MAE: ', mean_absolute_error(Y_valid, y_pred))\n    print('R2: ', r2_score(Y_valid, y_pred))\n\n    cat_prediction += best_model.predict(Pool(data=X_test))\n        \ncat_prediction /= n_fold\n\nprint('='*45)\nprint('CV mean MAE: {0:.4f}, std: {1:.4f}.'.format(np.mean(mae), np.std(mae)))\nprint('CV mean R2:  {0:.4f}, std: {1:.4f}.'.format(np.mean(r2), np.std(r2)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plotfig(best_model.predict(X), Y, 'Predicted vs. Actual responses for Catboost', max(Y) + 0.1*max(Y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train_features = train\n# train_targets = pd.DataFrame({'target':target})\n# test_features = test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# hyper_params = {\n#     'task': 'train',\n#     'boosting_type': 'gbdt',\n#     'objective': 'regression',\n#     'metric': ['mae'],\n#     'learning_rate': 0.067,\n#     'feature_fraction': 0.9,\n#     'subsample': 0.85,\n#     'subsample_freq': 2,\n#     'verbose': 500,\n#     \"max_depth\": -1,\n#     \"num_leaves\": 31,  \n#     \"max_bin\": 128,\n#     \"num_iterations\": 10000,\n#     'device': 'gpu',\n#     'gpu_platform_id': 0,\n#     'gpu_device_id': 0\n# }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# from sklearn.metrics import mean_absolute_error\n# import lightgbm as lgb\n# import math  \n# from sklearn.model_selection import KFold, StratifiedKFold\n\n# score = []\n\n# skf = KFold(n_splits = 5, shuffle=True, random_state=123)\n# skf.get_n_splits(train_features, train_targets)\n# oof_lgbm_df = pd.DataFrame()\n# predictions = pd.DataFrame(test_index)\n# x_test = test_features\n\n\n# for fold, (trn_idx, val_idx) in enumerate(skf.split(train_features, train_targets)):\n#     x_train, y_train = train_features.iloc[trn_idx], train_targets.iloc[trn_idx]['target']\n#     x_valid, y_valid = train_features.iloc[val_idx], train_targets.iloc[val_idx]['target']\n#     index = x_valid.index\n#     p_valid = 0\n#     yp = 0\n#     gbm = lgb.LGBMRegressor(**hyper_params)\n#     gbm.fit(x_train, y_train,\n#         eval_set=[(x_valid, y_valid)],\n#         eval_metric='mae',\n#         verbose = 500,\n#         early_stopping_rounds=100)\n#     score.append(mean_absolute_error(gbm.predict(x_valid), y_valid))\n#     yp += gbm.predict(x_test)\n#     fold_pred = pd.DataFrame({'ID': index,\n#                               'label':gbm.predict(x_valid)})\n#     oof_rfr_df = pd.concat([oof_lgbm_df, fold_pred], axis=0)\n#     predictions['fold{}'.format(fold+1)] = yp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# pred = (predictions['fold1'] + predictions['fold2'] + predictions['fold3'] + predictions['fold4'] + predictions['fold5'])/5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['segment_id'] = test_index\nsubmission['time_to_eruption'] = cat_prediction\nsubmission.to_csv('submission.csv', header=True, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}