{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U scikit-learn\n\nimport os\nimport time\n\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\n\nRANDOM_SEED = 111\n\nnp.random.seed(RANDOM_SEED)\n\nfrom datetime import datetime\nfrom numpy.random import default_rng\nrng = default_rng(RANDOM_SEED)\n\nimport matplotlib.pyplot as plt\n\nimport sklearn\nfrom sklearn.metrics import mean_squared_log_error, mean_squared_error, mean_squared_log_error\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score, accuracy_score, make_scorer\nfrom sklearn.preprocessing import OrdinalEncoder, MinMaxScaler, StandardScaler, OneHotEncoder, Binarizer, KBinsDiscretizer, QuantileTransformer, PolynomialFeatures, LabelEncoder\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, GridSearchCV, KFold, StratifiedKFold, StratifiedShuffleSplit, ShuffleSplit\nfrom sklearn.pipeline import Pipeline, FeatureUnion\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.linear_model import LinearRegression, Ridge, RidgeCV\nfrom sklearn.ensemble import StackingRegressor\n\nDS_DIR = '/kaggle/input/aug21-ds'\nINPUT_DIR = '/kaggle/input/tabular-playground-series-aug-2021'\nOUTPUT_DIR = './'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T03:18:23.860637Z","iopub.execute_input":"2022-07-29T03:18:23.861021Z","iopub.status.idle":"2022-07-29T03:18:38.648373Z","shell.execute_reply.started":"2022-07-29T03:18:23.860938Z","shell.execute_reply":"2022-07-29T03:18:38.647540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sklearn.__version__","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:18:38.649952Z","iopub.execute_input":"2022-07-29T03:18:38.650324Z","iopub.status.idle":"2022-07-29T03:18:38.657526Z","shell.execute_reply.started":"2022-07-29T03:18:38.650237Z","shell.execute_reply":"2022-07-29T03:18:38.656743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q h2o\n\nimport h2o\nfrom h2o.automl import H2OAutoML\nfrom h2o.sklearn import H2OAutoMLRegressor\nh2o.init()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:18:38.659355Z","iopub.execute_input":"2022-07-29T03:18:38.659999Z","iopub.status.idle":"2022-07-29T03:18:51.830097Z","shell.execute_reply.started":"2022-07-29T03:18:38.659958Z","shell.execute_reply":"2022-07-29T03:18:51.829201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load source data\n\nWe will split train dataset into the train (80%) and holdout (20%) to validate train results and stacking.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(INPUT_DIR, 'train.csv'), index_col='id')\ntest_df = pd.read_csv(os.path.join(INPUT_DIR,'test.csv'), index_col='id')\n\ntrain_df = train_df.sample(frac=1, random_state=RANDOM_SEED)\nholdout_size = train_df.shape[0]//5\nholdout_df = train_df.iloc[:holdout_size]\ntrain_df = train_df.iloc[holdout_size:]\nholdout_labels_df = holdout_df['loss']\nholdout_df.drop(columns='loss', inplace=True)\n\nlabels = train_df['loss']\ntrain_df.drop(columns='loss', inplace=True)\ntotal_df = train_df.append(holdout_df).append(test_df) \n\nprint('validate sample seed: ', holdout_df.iloc[100, :5])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:18:51.834866Z","iopub.execute_input":"2022-07-29T03:18:51.837003Z","iopub.status.idle":"2022-07-29T03:19:04.628396Z","shell.execute_reply.started":"2022-07-29T03:18:51.836957Z","shell.execute_reply":"2022-07-29T03:19:04.627500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat((total_df.min(), total_df.max(), total_df.mean(), total_df.std(), total_df.nunique()), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:19:04.629850Z","iopub.execute_input":"2022-07-29T03:19:04.630250Z","iopub.status.idle":"2022-07-29T03:19:06.452237Z","shell.execute_reply.started":"2022-07-29T03:19:04.630210Z","shell.execute_reply":"2022-07-29T03:19:06.451403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build H2O model\n\nThe built-in H2O stacking algorithm reach as much as **7.87097** in private score.<br/>\nTo improve the scores, increase the `max_runtime_secs` to 1h.","metadata":{}},{"cell_type":"code","source":"def RMSE(y_true, y_pred, **kwargs):\n    return np.sqrt(mean_squared_error(y_true, y_pred, **kwargs))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:19:06.453489Z","iopub.execute_input":"2022-07-29T03:19:06.454008Z","iopub.status.idle":"2022-07-29T03:19:06.458446Z","shell.execute_reply.started":"2022-07-29T03:19:06.453968Z","shell.execute_reply":"2022-07-29T03:19:06.457367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_frame = h2o.H2OFrame(train_df)\nholdout_frame = h2o.H2OFrame(holdout_df)\ntest_frame = h2o.H2OFrame(test_df)\n\nmodel = H2OAutoMLRegressor(seed=RANDOM_SEED, max_runtime_secs=600, nfolds=5, stopping_metric='RMSE', sort_metric='RMSE', \n                           stopping_rounds=10, verbosity='warn')\n\nmodel.fit(train_frame, labels.values)\ntest_pred = np.squeeze(model.predict(test_frame).as_data_frame().values)\ntrain_pred = np.squeeze(model.predict(train_frame).as_data_frame().values)\nholdout_pred = np.squeeze(model.predict(holdout_frame).as_data_frame().values)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2022-07-29T03:19:06.489680Z","iopub.execute_input":"2022-07-29T03:19:06.489992Z","iopub.status.idle":"2022-07-29T03:30:48.265910Z","shell.execute_reply.started":"2022-07-29T03:19:06.489959Z","shell.execute_reply":"2022-07-29T03:30:48.265016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_train = RMSE(train_pred, labels.values)\nscore_holdout = RMSE(holdout_pred, holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:30:48.267984Z","iopub.execute_input":"2022-07-29T03:30:48.268418Z","iopub.status.idle":"2022-07-29T03:30:48.282713Z","shell.execute_reply.started":"2022-07-29T03:30:48.268376Z","shell.execute_reply":"2022-07-29T03:30:48.281765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.estimator.leaderboard.as_data_frame().iloc[:12]","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:53:27.177778Z","iopub.execute_input":"2022-07-29T03:53:27.178098Z","iopub.status.idle":"2022-07-29T03:53:27.203387Z","shell.execute_reply.started":"2022-07-29T03:53:27.178068Z","shell.execute_reply":"2022-07-29T03:53:27.202548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#h2o.get_model(model.estimator.get_best_model().metalearner().model_id)\n\nleaderbrd = model.estimator.leaderboard.as_data_frame().iloc[2:12]\nleaderbrd = leaderbrd['model_id'].values\n\ntest_ds_pred, train_ds_pred, holdout_ds_pred = [], [], []\nfor mdl in leaderbrd:\n    mdl = h2o.get_model(mdl)\n    test_ds_pred.append(np.squeeze(mdl.predict(test_frame).as_data_frame().values))\n    train_ds_pred.append(np.squeeze(mdl.predict(train_frame).as_data_frame().values))\n    holdout_ds_pred.append(np.squeeze(mdl.predict(holdout_frame).as_data_frame().values))\n\ntest_ds_pred = np.vstack(test_ds_pred).T\ntrain_ds_pred = np.vstack(train_ds_pred).T\nholdout_ds_pred = np.vstack(holdout_ds_pred).T","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-29T03:30:48.311479Z","iopub.execute_input":"2022-07-29T03:30:48.311894Z","iopub.status.idle":"2022-07-29T03:31:08.598712Z","shell.execute_reply.started":"2022-07-29T03:30:48.311854Z","shell.execute_reply":"2022-07-29T03:31:08.597831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save new or load existing predicted data","metadata":{}},{"cell_type":"code","source":"%%script echo skipping\n\noutput_res = pd.DataFrame(data=test_ds_pred)\noutput_res.to_csv('test_pred.csv', index=False)\n\noutput_res = pd.DataFrame(data=train_ds_pred)\noutput_res.to_csv('train_pred.csv', index=False)\n\noutput_res = pd.DataFrame(data=holdout_ds_pred)\noutput_res.to_csv('holdout_pred.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:08.600423Z","iopub.execute_input":"2022-07-29T03:31:08.600936Z","iopub.status.idle":"2022-07-29T03:31:08.626433Z","shell.execute_reply.started":"2022-07-29T03:31:08.600898Z","shell.execute_reply":"2022-07-29T03:31:08.625633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%script echo skipping\n\ntest_ds_pred = pd.read_csv(os.path.join(DS_DIR,'test_pred.csv')).values\ntrain_ds_pred = pd.read_csv(os.path.join(DS_DIR, 'train_pred.csv')).values\nholdout_ds_pred = pd.read_csv(os.path.join(DS_DIR, 'holdout_pred.csv')).values\n\ntrain_labels = labels.values\nholdout_labels = holdout_labels_df.values","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:08.628163Z","iopub.execute_input":"2022-07-29T03:31:08.628606Z","iopub.status.idle":"2022-07-29T03:31:08.652074Z","shell.execute_reply.started":"2022-07-29T03:31:08.628529Z","shell.execute_reply":"2022-07-29T03:31:08.651043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Do manual stacking and blending\nWe will do and compare stacking among the best 10 sub-models from the same H2O model\n\n|stacking type|train score|validation score\n|---|---|---\n|Original stacking|7.68904|7.85766\n|Take the best fold in holdout dataset (no stacking)|7.70764|7.86561\n|Mean/median among numeric float label values|7.8829|7.89459\n|Voting among discreet label values|7.94684|7.91884\n|Best 3 models weighted average against holdout labels|7.72745|7.85960\n|Add a normally-distributed noise to original stacking|7.68913|7.85757\n|2nd H2O stacking with predicted values from holdout dataset|7.75347|7.85057\n|H2O the original stacking with 10% test pseudolabels|7.68497|7.85747\n|H2O recursive chaining with pseudolabels|7.72745|7.85960","metadata":{}},{"cell_type":"code","source":"#1) Take the best model in holdout dataset (no stacking)\n\nbest_holdout_score, scores, best_idx = 10, [], -1\nfor i in range(train_ds_pred.shape[1]):\n    score_train = RMSE(train_ds_pred.T[i], labels.values)\n    score_holdout = RMSE(holdout_ds_pred.T[i], holdout_labels_df.values)\n    print(score_train, score_holdout)\n    scores.append(score_holdout)\n    if best_holdout_score > score_holdout:\n        best_holdout_score = score_holdout\n        best_idx = i\n\nscores = np.argsort(scores)\n\nprint('\\n', 'index with best score:', best_idx, ', score ranks:', scores)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:08.655410Z","iopub.execute_input":"2022-07-29T03:31:08.655705Z","iopub.status.idle":"2022-07-29T03:31:08.726396Z","shell.execute_reply.started":"2022-07-29T03:31:08.655676Z","shell.execute_reply":"2022-07-29T03:31:08.725627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2) Mean/median among numeric float label values\n\nscore_train = RMSE(np.mean(train_ds_pred, 1), labels.values)\nscore_holdout = RMSE(np.mean(holdout_ds_pred, 1), holdout_labels_df.values)\nprint(score_train, score_holdout)\n\nscore_train = RMSE(np.median(train_ds_pred, 1), labels.values)\nscore_holdout = RMSE(np.median(holdout_ds_pred, 1), holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:08.727577Z","iopub.execute_input":"2022-07-29T03:31:08.727911Z","iopub.status.idle":"2022-07-29T03:31:08.815318Z","shell.execute_reply.started":"2022-07-29T03:31:08.727874Z","shell.execute_reply":"2022-07-29T03:31:08.814298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3) Voting among discreet label values\n\nscore_train = RMSE(np.squeeze(stats.mode(np.round(train_ds_pred), 1).mode), labels.values)\nscore_holdout = RMSE(np.squeeze(stats.mode(np.round(holdout_ds_pred), 1).mode), holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:08.816792Z","iopub.execute_input":"2022-07-29T03:31:08.817150Z","iopub.status.idle":"2022-07-29T03:31:17.321823Z","shell.execute_reply.started":"2022-07-29T03:31:08.817113Z","shell.execute_reply":"2022-07-29T03:31:17.320936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#4) Best weighted average against holdout labels\n#7.877126121980474 (0.075, 0.8, 0.125) - Private Score: 7.87690\n#7.885931889115389 (0.0875, 0.8, 0.1125) - Private Score: 7.87679\n#7.859609548367599 (0.73125, 0.2375, 0.03125) [0 2 1]\n\n\nimport itertools\nfrom tqdm.notebook import tqdm\n\ndef minimize_weights(cols, n, maxv):\n    size = len(cols)\n    best_score = 100\n    best_weights = []\n    perf = np.array([0]*size)\n    score = 100\n\n    linspace = np.round(np.linspace(0, maxv, round(n*maxv*10)+1).tolist(), 8)\n    print(linspace[:50])\n    print(cols)\n\n    for x in itertools.product(linspace, repeat=size): \n        if sum(x) == 1:\n            score = mean_squared_error(np.average(holdout_ds_pred.T[cols], 0, x), holdout_labels_df.values)\n        if score < best_score:\n            best_score = score\n            best_weights = x\n            perf = perf + x\n            #print(np.sqrt(score), x)\n\n    cols = cols[perf.argsort()]\n    print(np.sqrt(best_score), best_weights, cols)\n    return cols, best_weights\n\n#full flow for selecting best features:\n#cols = scores\n#cols, best_weights = minimize_weights(cols[:12], 0.5, 0.8)\n#cols, best_weights = minimize_weights(cols[-6:], 2, 0.8)\n#cols, best_weights = minimize_weights(cols[-4:], 4, 0.8)\n#cols, best_weights = minimize_weights(cols[-3:], 8, 0.8)\n\n#quick flow:\ncols, best_weights = minimize_weights(np.array([0,1,2]), 16, 0.8)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:17:38.962708Z","iopub.execute_input":"2022-07-29T05:17:38.963034Z","iopub.status.idle":"2022-07-29T05:17:50.287337Z","shell.execute_reply.started":"2022-07-29T05:17:38.963005Z","shell.execute_reply":"2022-07-29T05:17:50.286433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = np.array([0,1,2])\nscore_train = RMSE(np.average(train_ds_pred.T[cols], 0, best_weights), labels.values)\nscore_holdout = RMSE(np.average(holdout_ds_pred.T[cols], 0, best_weights), holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:19:38.664779Z","iopub.execute_input":"2022-07-29T05:19:38.665109Z","iopub.status.idle":"2022-07-29T05:19:38.680223Z","shell.execute_reply.started":"2022-07-29T05:19:38.665077Z","shell.execute_reply":"2022-07-29T05:19:38.679093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#5) Add a normally-distributed noise to existing predictions\n\ntrain_pred_noise = [x + np.random.normal(0, 0.05) for x in train_pred]\nholdout_pred_noise = [x + np.random.normal(0, 0.05) for x in holdout_pred]\n\nscore_train = RMSE(train_pred_noise, labels.values)\nscore_holdout = RMSE(holdout_pred_noise, holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:04:09.689224Z","iopub.execute_input":"2022-07-29T04:04:09.689549Z","iopub.status.idle":"2022-07-29T04:04:10.974791Z","shell.execute_reply.started":"2022-07-29T04:04:09.689517Z","shell.execute_reply":"2022-07-29T04:04:10.973909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#6) H2O stacking with predicted values from holdout dataset\n\nholdout_stack_frame = h2o.H2OFrame(holdout_ds_pred)\ntrain_stack_frame = h2o.H2OFrame(train_ds_pred)\n\nmodel = H2OAutoMLRegressor(seed=RANDOM_SEED, max_runtime_secs=600, nfolds=5, stopping_metric='RMSE', sort_metric='RMSE', \n                           stopping_rounds=10, verbosity='warn')\n\nmodel.fit(holdout_stack_frame, holdout_labels_df.values)\ntrain_stack_pred = np.squeeze(model.predict(train_stack_frame).as_data_frame().values)\nholdout_stack_pred = np.squeeze(model.predict(holdout_stack_frame).as_data_frame().values)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:04:48.341240Z","iopub.execute_input":"2022-07-29T04:04:48.341588Z","iopub.status.idle":"2022-07-29T04:14:43.617304Z","shell.execute_reply.started":"2022-07-29T04:04:48.341537Z","shell.execute_reply":"2022-07-29T04:14:43.616359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_train = RMSE(train_stack_pred, labels.values)\nscore_holdout = RMSE(holdout_stack_pred, holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:19:44.401525Z","iopub.execute_input":"2022-07-29T04:19:44.401895Z","iopub.status.idle":"2022-07-29T04:19:44.412948Z","shell.execute_reply.started":"2022-07-29T04:19:44.401864Z","shell.execute_reply":"2022-07-29T04:19:44.411888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#7) H2O default stacking with 10% test pseudolabels\n\ntest_pseudo_df = test_df.copy()\ntest_pseudo_df['loss'] = test_pred\ntest_pseudo_df = test_pseudo_df.sample(frac=1, random_state=RANDOM_SEED)\ntest_pseudo_df = test_pseudo_df.iloc[:test_pseudo_df.shape[0]//10]\n\ntrain_pseudo_df = train_df.copy()\ntrain_pseudo_df['loss'] = labels\ntrain_pseudo_df = train_pseudo_df.append(test_pseudo_df)\ntrain_pseudo_labels = train_pseudo_df['loss']\ntrain_pseudo_df.drop(columns='loss', inplace=True)\n\nmodel = H2OAutoMLRegressor(seed=RANDOM_SEED, max_runtime_secs=600, nfolds=5, stopping_metric='RMSE', sort_metric='RMSE', \n                           stopping_rounds=10, verbosity='warn')\n\nmodel.fit(train_pseudo_df.values, train_pseudo_labels.values)\ntrain_pred2 = model.predict(train_df.values)\nholdout_pred2 = model.predict(holdout_df.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:37:30.729046Z","iopub.execute_input":"2022-07-29T04:37:30.729408Z","iopub.status.idle":"2022-07-29T04:48:58.578540Z","shell.execute_reply.started":"2022-07-29T04:37:30.729376Z","shell.execute_reply":"2022-07-29T04:48:58.577709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_train = RMSE(train_pred2, labels.values)\nscore_holdout = RMSE(holdout_pred2, holdout_labels_df.values)\nprint(score_train, score_holdout)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:49:02.438178Z","iopub.execute_input":"2022-07-29T04:49:02.438514Z","iopub.status.idle":"2022-07-29T04:49:02.449790Z","shell.execute_reply.started":"2022-07-29T04:49:02.438481Z","shell.execute_reply":"2022-07-29T04:49:02.448661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#8) H2O recursive chaining with pseudolabels\n#The algorithm is taken from:\n#https://www.kaggle.com/c/tabular-playground-series-aug-2021/discussion/270051\n\nmodel = H2OAutoMLRegressor(seed=RANDOM_SEED, max_runtime_secs=300, nfolds=5, stopping_metric='RMSE', sort_metric='RMSE', \n                           stopping_rounds=10, verbosity='warn')\n\ndef recursive_train(test_pred, learning_rate = 0.2, output_res=None):\n    model.fit(test_frame, test_pred)\n    train_pred = np.squeeze(model.predict(train_frame).as_data_frame().values)\n    holdout_pred = np.squeeze(model.predict(holdout_frame).as_data_frame().values)\n    print('train loss:', RMSE(train_pred, labels.values), 'holdout loss:', RMSE(holdout_pred, holdout_labels_df.values))\n\n    model.fit(train_frame, labels.values - train_pred)\n    error_prediction = np.squeeze(model.predict(test_frame).as_data_frame().values)\n\n    test_pred = test_pred + (error_prediction * learning_rate)\n    if output_res:\n        output_res['loss'] = test_pred\n        output_res.to_csv('submission.csv', index=False)\n\n    return test_pred\n\nmodel.fit(train_frame, labels.values)\ntest_pred = np.squeeze(model.predict(test_frame).as_data_frame().values)\ntrain_pred = np.squeeze(model.predict(train_frame).as_data_frame().values)\nholdout_pred = np.squeeze(model.predict(holdout_frame).as_data_frame().values)\nprint('train loss:', RMSE(train_pred, labels.values), 'holdout loss:', RMSE(holdout_pred, holdout_labels_df.values))\n\ntest_pred_new = test_pred\n\nfor x in range(10):\n    test_pred_new = recursive_train(test_pred_new)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:20:14.670012Z","iopub.execute_input":"2022-07-29T05:20:14.670350Z","iopub.status.idle":"2022-07-29T07:09:32.492744Z","shell.execute_reply.started":"2022-07-29T05:20:14.670319Z","shell.execute_reply":"2022-07-29T07:09:32.491822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save submission data","metadata":{}},{"cell_type":"code","source":"output_res = pd.DataFrame(index=test_df.index, data={'id':test_df.index})\noutput_res['loss'] = test_pred_new\noutput_res.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T03:31:17.470691Z","iopub.status.idle":"2022-07-29T03:31:17.471399Z"},"trusted":true},"execution_count":null,"outputs":[]}]}