{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"A lot of the code here was inspired from Abhishek Thakur's [notebook on stacking](https://www.kaggle.com/code/abhishek/competition-part-6-stacking). Make sure to check that out as well.","metadata":{}},{"cell_type":"markdown","source":"# Introduction\n\nStacking is an ensemble machine learning algorithm that uses meta-learning and is often the final step in tabular-data competitions. It learns how to best combine the predictions from multiple well-performing models on a classification or regression task and make predictions that generally have better performance.\n\nBy studying this tutorial you will learn how to perform stacking using TensorFlow Decision Forests. This notebook is a continuation of the [notebook on blending](https://www.kaggle.com/code/rishirajacharya/gsoc-2022-tf-df-tuning-blending). The predictions on training and test data from multiple models have been imported as a dataset here.","metadata":{}},{"cell_type":"markdown","source":"# Installing TensorFlow Decision Forests","metadata":{}},{"cell_type":"code","source":"# Display only the messages with ERROR, CRITICAL log levels\n!pip install tensorflow_decision_forests -U -qq","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.metrics import mean_squared_error\nimport tensorflow as tf\nimport tensorflow_decision_forests as tfdf\nfrom sklearn.linear_model import LinearRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dataset of predictions on test & out-of-fold training set","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tpsfeb21-folds/train_folds.csv\")\ndf_test = pd.read_csv(\"../input/tabular-playground-series-feb-2021/test.csv\")\n\ndf1 = pd.read_csv(\"../input/stackingtpsfeb21/train_pred_1.csv\")\ndf1.columns = [\"id\", \"pred_1\"]\ndf2 = pd.read_csv(\"../input/stackingtpsfeb21/train_pred_2.csv\")\ndf2.columns = [\"id\", \"pred_2\"]\ndf3 = pd.read_csv(\"../input/stackingtpsfeb21/train_pred_3.csv\")\ndf3.columns = [\"id\", \"pred_3\"]\ndf4 = pd.read_csv(\"../input/stackingtpsfeb21/train_pred_4.csv\")\ndf4.columns = [\"id\", \"pred_4\"]\n\ndf_test1 = pd.read_csv(\"../input/stackingtpsfeb21/test_pred_1.csv\")\ndf_test1.columns = [\"id\", \"pred_1\"]\ndf_test2 = pd.read_csv(\"../input/stackingtpsfeb21/test_pred_2.csv\")\ndf_test2.columns = [\"id\", \"pred_2\"]\ndf_test3 = pd.read_csv(\"../input/stackingtpsfeb21/test_pred_3.csv\")\ndf_test3.columns = [\"id\", \"pred_3\"]\ndf_test4 = pd.read_csv(\"../input/stackingtpsfeb21/test_pred_4.csv\")\ndf_test4.columns = [\"id\", \"pred_4\"]\n\ndf = df.merge(df1, on=\"id\", how=\"left\")\ndf = df.merge(df2, on=\"id\", how=\"left\")\ndf = df.merge(df3, on=\"id\", how=\"left\")\ndf = df.merge(df4, on=\"id\", how=\"left\")\n\ndf_test = df_test.merge(df_test1, on=\"id\", how=\"left\")\ndf_test = df_test.merge(df_test2, on=\"id\", how=\"left\")\ndf_test = df_test.merge(df_test3, on=\"id\", how=\"left\")\ndf_test = df_test.merge(df_test4, on=\"id\", how=\"left\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to a tf.Dataset & fit a TF-DF Gradient Boosted Trees model\n\nFit it on the Level 0 model predictions from the notebook on blending to generate first Level 1 predictions.","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"../input/tabular-playground-series-feb-2021/sample_submission.csv\")\nuseful_features = [\"pred_1\", \"pred_2\", \"pred_3\", \"pred_4\", \"target\"]\nuseful_features_test = [\"pred_1\", \"pred_2\", \"pred_3\", \"pred_4\"]\ndf_test = df_test[useful_features_test]\n\nfinal_test_predictions = []\nfinal_valid_predictions = {}\nscores = []\nfor fold in range(5):\n    xtrain =  df[df.kfold != fold].reset_index(drop=True)\n    xvalid = df[df.kfold == fold].reset_index(drop=True)\n    xtest = df_test.copy()\n\n    valid_ids = xvalid.id.values.tolist()\n\n    ytrain = xtrain.target\n    yvalid = xvalid.target\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n    \n    xtrain_ds = tfdf.keras.pd_dataframe_to_tf_dataset(xtrain, label=\"target\", task=tfdf.keras.Task.REGRESSION)\n    xvalid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(xvalid, label=\"target\", task=tfdf.keras.Task.REGRESSION)\n    xtest_ds = tfdf.keras.pd_dataframe_to_tf_dataset(xtest, task=tfdf.keras.Task.REGRESSION)\n    \n    model = tfdf.keras.GradientBoostedTreesModel(task=tfdf.keras.Task.REGRESSION)\n    model.fit(x=xtrain_ds)\n    \n    preds_valid = model.predict(xvalid_ds)\n    test_preds = model.predict(xtest_ds)\n    final_test_predictions.append(test_preds)\n    final_valid_predictions.update(dict(zip(valid_ids, preds_valid)))\n    rmse = mean_squared_error(yvalid, preds_valid, squared=False)\n    print(fold, rmse)\n    scores.append(rmse)\n\nprint(np.mean(scores), np.std(scores))\nfinal_valid_predictions = pd.DataFrame.from_dict(final_valid_predictions, orient=\"index\").reset_index()\nfinal_valid_predictions.columns = [\"id\", \"pred_1\"]\nfinal_valid_predictions.to_csv(\"level1_train_pred_1.csv\", index=False)\n\nsample_submission.target = np.mean(np.column_stack(final_test_predictions), axis=1)\nsample_submission.columns = [\"id\", \"pred_1\"]\nsample_submission.to_csv(\"level1_test_pred_1.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit a Random Forest Regressor model\n\nFit it on the Level 0 model predictions from the notebook on blending to generate second Level 1 predictions.","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"../input/tabular-playground-series-feb-2021/sample_submission.csv\")\nuseful_features = [\"pred_1\", \"pred_2\", \"pred_3\", \"pred_4\"]\ndf_test = df_test[useful_features]\n\nfinal_test_predictions = []\nfinal_valid_predictions = {}\nscores = []\nfor fold in range(5):\n    xtrain =  df[df.kfold != fold].reset_index(drop=True)\n    xvalid = df[df.kfold == fold].reset_index(drop=True)\n    xtest = df_test.copy()\n\n    valid_ids = xvalid.id.values.tolist()\n\n    ytrain = xtrain.target\n    yvalid = xvalid.target\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n    \n    model = RandomForestRegressor(n_estimators=500, n_jobs=-1, max_depth=3)\n    model.fit(xtrain, ytrain)\n    preds_valid = model.predict(xvalid)\n    test_preds = model.predict(xtest)\n    final_test_predictions.append(test_preds)\n    final_valid_predictions.update(dict(zip(valid_ids, preds_valid)))\n    rmse = mean_squared_error(yvalid, preds_valid, squared=False)\n    print(fold, rmse)\n    scores.append(rmse)\n\nprint(np.mean(scores), np.std(scores))\nfinal_valid_predictions = pd.DataFrame.from_dict(final_valid_predictions, orient=\"index\").reset_index()\nfinal_valid_predictions.columns = [\"id\", \"pred_2\"]\nfinal_valid_predictions.to_csv(\"level1_train_pred_2.csv\", index=False)\n\nsample_submission.target = np.mean(np.column_stack(final_test_predictions), axis=1)\nsample_submission.columns = [\"id\", \"pred_2\"]\nsample_submission.to_csv(\"level1_test_pred_2.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit a Gradient Boosting Regressor model\n\nFit it on the Level 0 model predictions from the notebook on blending to generate third Level 1 predictions.","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"../input/tabular-playground-series-feb-2021/sample_submission.csv\")\nuseful_features = [\"pred_1\", \"pred_2\", \"pred_3\", \"pred_4\"]\ndf_test = df_test[useful_features]\n\nfinal_test_predictions = []\nfinal_valid_predictions = {}\nscores = []\nfor fold in range(5):\n    xtrain =  df[df.kfold != fold].reset_index(drop=True)\n    xvalid = df[df.kfold == fold].reset_index(drop=True)\n    xtest = df_test.copy()\n\n    valid_ids = xvalid.id.values.tolist()\n\n    ytrain = xtrain.target\n    yvalid = xvalid.target\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n    \n    model = GradientBoostingRegressor(n_estimators=500, max_depth=3)\n    model.fit(xtrain, ytrain)\n    preds_valid = model.predict(xvalid)\n    test_preds = model.predict(xtest)\n    final_test_predictions.append(test_preds)\n    final_valid_predictions.update(dict(zip(valid_ids, preds_valid)))\n    rmse = mean_squared_error(yvalid, preds_valid, squared=False)\n    print(fold, rmse)\n    scores.append(rmse)\n\nprint(np.mean(scores), np.std(scores))\nfinal_valid_predictions = pd.DataFrame.from_dict(final_valid_predictions, orient=\"index\").reset_index()\nfinal_valid_predictions.columns = [\"id\", \"pred_3\"]\nfinal_valid_predictions.to_csv(\"level1_train_pred_3.csv\", index=False)\n\nsample_submission.target = np.mean(np.column_stack(final_test_predictions), axis=1)\nsample_submission.columns = [\"id\", \"pred_3\"]\nsample_submission.to_csv(\"level1_test_pred_3.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge the predictions from the three models into a single dataframe.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tpsfeb21-folds/train_folds.csv\")\ndf_test = pd.read_csv(\"../input/tabular-playground-series-feb-2021/test.csv\")\nsample_submission = pd.read_csv(\"../input/tabular-playground-series-feb-2021/sample_submission.csv\")\n\ndf1 = pd.read_csv(\"level1_train_pred_1.csv\")\ndf2 = pd.read_csv(\"level1_train_pred_2.csv\")\ndf3 = pd.read_csv(\"level1_train_pred_3.csv\")\n\ndf_test1 = pd.read_csv(\"level1_test_pred_1.csv\")\ndf_test2 = pd.read_csv(\"level1_test_pred_2.csv\")\ndf_test3 = pd.read_csv(\"level1_test_pred_3.csv\")\n\ndf = df.merge(df1, on=\"id\", how=\"left\")\ndf = df.merge(df2, on=\"id\", how=\"left\")\ndf = df.merge(df3, on=\"id\", how=\"left\")\n\ndf_test = df_test.merge(df_test1, on=\"id\", how=\"left\")\ndf_test = df_test.merge(df_test2, on=\"id\", how=\"left\")\ndf_test = df_test.merge(df_test3, on=\"id\", how=\"left\")\n\ndf.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fit a Linear Regression model over the Level 1 predictions that finds the coefficients. This is generally a smarter way to blend than simply averaging over the predictions.","metadata":{}},{"cell_type":"code","source":"useful_features = [\"pred_1\", \"pred_2\", \"pred_3\"]\ndf_test = df_test[useful_features]\n\nfinal_predictions = []\nscores = []\nfor fold in range(5):\n    xtrain =  df[df.kfold != fold].reset_index(drop=True)\n    xvalid = df[df.kfold == fold].reset_index(drop=True)\n    xtest = df_test.copy()\n\n    ytrain = xtrain.target\n    yvalid = xvalid.target\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n    \n    model = LinearRegression()\n    model.fit(xtrain, ytrain)\n    \n    preds_valid = model.predict(xvalid)\n    test_preds = model.predict(xtest)\n    final_predictions.append(test_preds)\n    rmse = mean_squared_error(yvalid, preds_valid, squared=False)\n    print(fold, rmse)\n    scores.append(rmse)\n\nprint(np.mean(scores), np.std(scores))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make submission","metadata":{}},{"cell_type":"code","source":"sample_submission.target = np.mean(np.column_stack(final_predictions), axis=1)\nsample_submission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}