{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is a catboost baseline for prediction time to eruption using the generated dataset with TSFresh library. \nI used the EfficientFCParameters() parameter that results in 773 features generation for each time series. Both train and test datasets were processing during 5 days on 30 threads of Ryzen 1950X. Finally, we have 7730 features for each observation. The dataset is uploaded as ingv-tsfresh-7730. Let's improve the scores :) "},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2020-10-13T17:35:57.667616Z","iopub.status.busy":"2020-10-13T17:35:57.666853Z","iopub.status.idle":"2020-10-13T17:35:57.670266Z","shell.execute_reply":"2020-10-13T17:35:57.669645Z"},"papermill":{"duration":0.021333,"end_time":"2020-10-13T17:35:57.670401","exception":false,"start_time":"2020-10-13T17:35:57.649068","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nimport pandas as pd\n\nfrom time import time\nfrom time import ctime\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom tqdm import tqdm_notebook\nfrom tqdm import tqdm\n\nimport joblib\nfrom joblib import Parallel, delayed\nimport multiprocessing\nnum_cores = multiprocessing.cpu_count()-1\n\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import r2_score\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import KFold\nimport matplotlib.pyplot as plt\n\ndef plotfig (ypred, yactual, strtitle, y_max):\n    plt.scatter(ypred, yactual.values.ravel())\n    plt.title(strtitle)\n    plt.plot([(0, 0), (y_max, y_max)], [(0, 0), (y_max, y_max)])\n    plt.xlim(0, y_max)\n    plt.ylim(0, y_max)\n    plt.xlabel('Predicted', fontsize=12)\n    plt.ylabel('Actual', fontsize=12)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.011327,"end_time":"2020-10-13T17:35:58.926548","exception":false,"start_time":"2020-10-13T17:35:58.915221","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### Load prepared dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/ingv-tsfresh-7730/train.csv', sep = ';')\ntrain.set_index('Unnamed: 0', inplace = True)\ntest = pd.read_csv('../input/ingv-tsfresh-7730/test.csv', sep = ';')\ntest.set_index('Unnamed: 0', inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()     ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Or create it using the TSFresh lib"},{"metadata":{"trusted":true},"cell_type":"code","source":"# from tsfresh import select_features\n# from tsfresh.utilities.dataframe_functions import impute\n# from tsfresh import extract_relevant_features\n# from tsfresh.feature_extraction import settings, extract_features, MinimalFCParameters, EfficientFCParameters, ComprehensiveFCParameters\n\n\n# def features_generator(path_to_file):\n#     signals = pd.read_csv(path_to_file)\n#     seg = int(path_to_file.split('/')[-1].split('.')[0])\n#     signals['segment_id'] = seg\n    \n#     sel = signals.fillna(0).astype(bool).sum(axis=0) / 60001 > 0.5\n#     signals = signals.fillna(0).loc[:,sel]\n\n#     extracted_features = extract_features(signals.iloc[:,:], \n#                                           column_id = 'segment_id', \n#                                           default_fc_parameters=EfficientFCParameters(),\n#                                           n_jobs = 0,\n#                                           disable_progressbar = False,\n#                                           chunksize = None,\n#                                          )\n#     return extracted_features\n\n# %%time\n# train_path_to_signals = 'G:/Kaggle/INGV_Volcanic_Eruption_Prediction/raw_data/predict-volcanic-eruptions-ingv-oe/train/'\n# train_files_list = [os.path.join(train_path_to_signals, file) for file in os.listdir(train_path_to_signals)]\n# rows = Parallel(n_jobs=30)(delayed(features_generator)(ex) for ex in tqdm_notebook(train_files_list[:]))  \n# train_set = pd.concat(rows, axis=0)\n\n# test_path_to_signals = 'G:/Kaggle/INGV_Volcanic_Eruption_Prediction/raw_data/predict-volcanic-eruptions-ingv-oe/test/'\n# test_files_list = [os.path.join(test_path_to_signals, file) for file in os.listdir(test_path_to_signals)]\n# rows = Parallel(n_jobs=30)(delayed(features_generator)(ex) for ex in tqdm_notebook(test_files_list[:]))  \n# test_set = pd.concat(rows, axis=0)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":true},"cell_type":"code","source":"%%time\n\nY = train['time_to_eruption']\nX = train.drop(['time_to_eruption'], axis = 1)\nX_test = test\n\nn_fold = 5\ncv = KFold(n_splits=n_fold, shuffle=True, random_state=42)\n\noof = np.zeros(len(X))\ncat_prediction = np.zeros(len(X_test))\nmae, r2 = [], []\n\nPARAMS = {\n    \n             'random_seed': 42,\n             'eval_metric': 'MAE',\n#              'iterations': 100,\n             'task_type': 'GPU',\n\n        }\n\nfor fold_n, (train_index, valid_index) in enumerate(cv.split(X)):\n    print('\\nFold', fold_n, 'started at', ctime())\n\n    X_train = X.iloc[train_index,:]\n    X_valid = X.iloc[valid_index,:]\n    \n    Y_train = Y.iloc[train_index]\n    Y_valid = Y.iloc[valid_index]\n          \n    best_model = CatBoostRegressor(**PARAMS, thread_count = -1)  \n    \n    train_dataset = Pool(data=X_train,\n                     label=Y_train,\n                     )\n    \n    eval_dataset = Pool(data=X_valid,\n                    label=Y_valid,\n                    )\n    \n    best_model.fit(train_dataset,\n              use_best_model=True,\n              verbose = False,\n              plot = True,\n              eval_set=eval_dataset,\n              early_stopping_rounds=100)\n\n   \n    y_pred = best_model.predict(Pool(data=X_valid))\n\n    mae.append(mean_absolute_error(Y_valid, y_pred))\n    r2.append(r2_score(Y_valid, y_pred))\n\n    print('MAE: ', mean_absolute_error(Y_valid, y_pred))\n    print('R2: ', r2_score(Y_valid, y_pred))\n\n    cat_prediction += best_model.predict(Pool(data=X_test))\n        \ncat_prediction /= n_fold\n\nprint('='*45)\nprint('CV mean MAE: {0:.4f}, std: {1:.4f}.'.format(np.mean(mae), np.std(mae)))\nprint('CV mean R2:  {0:.4f}, std: {1:.4f}.'.format(np.mean(r2), np.std(r2)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotfig(best_model.predict(X), Y, 'Predicted vs. Actual responses for Catboost', max(Y) + 0.1*max(Y))","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.068863,"end_time":"2020-10-13T23:08:42.229149","exception":false,"start_time":"2020-10-13T23:08:42.160286","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}