{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"As we approach the end of the competition, I'm sure we are all trying to get our ducks (models) in a row. In this notebook, we will go over some common techniques to building ensembles. The requirement to perform this reliably is that all the models in the ensemble need to be built using the same folds (this can be achieved by fixing the random seed when doing KFold splits). This is to ensure that each OOF prediction was made with the same data available to it\n\nI have used OOF from @cdeotte's XGBoost starter notebook as well as OOF predictions from a couple of models I have built. Just for illustrative purposes\n\nThe following techniques are shown below:\n\n1. Simple Averaging\n2. Weighted Averaging\n3. Rank Averaging\n4. Weighted Rank Averaging\n5. Hill climbing\n6. Weight Optimization\n\nOnce you find the best combination which works for the OOF predictions. Combine the test predictions in the same way and submit! Good luck to everyone :)\n\n\nCredits:\n\n- https://www.youtube.com/watch?v=TuIgtitqJho&t=3880s\n- https://www.kaggle.com/code/cdeotte/forward-selection-oof-ensemble-0-942-private/notebook","metadata":{}},{"cell_type":"markdown","source":"# Load libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:12.724947Z","iopub.execute_input":"2022-08-01T21:54:12.725435Z","iopub.status.idle":"2022-08-01T21:54:12.890042Z","shell.execute_reply.started":"2022-08-01T21:54:12.725338Z","shell.execute_reply":"2022-08-01T21:54:12.888775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = [\n            '../input/xgboost-starter-0-793/oof_xgb_v1.csv',\n            '../input/amexoofspublic/oof_xgb_v484_seed42.csv',\n            '../input/amexoofspublic/oof_xgb_v513_seed42.csv'\n        ]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:13.272727Z","iopub.execute_input":"2022-08-01T21:54:13.273168Z","iopub.status.idle":"2022-08-01T21:54:13.278925Z","shell.execute_reply.started":"2022-08-01T21:54:13.273131Z","shell.execute_reply":"2022-08-01T21:54:13.277416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df = pd.DataFrame()\n\nfor i, path in enumerate(paths):\n    \n    # Find filename\n    start = path.rfind('/')+1\n    end = path.find('.csv')\n    fname = path[start:end]\n    \n    print(path)\n    \n    temp = pd.read_csv(path)\n    \n    # Set column name of oof\n    temp.rename(columns={'oof_pred': f'{fname}'}, inplace=True)\n    \n    # Drop redundant target column\n    if i != 0:\n        temp.drop(columns=['target'], inplace=True)\n    \n    # Join to main file\n    if i == 0:\n        oof_df = temp\n    else:\n        oof_df = pd.merge(oof_df, temp, on=\"customer_ID\", how=\"left\")\n    \n\noof_df.shape","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T21:54:14.606576Z","iopub.execute_input":"2022-08-01T21:54:14.607749Z","iopub.status.idle":"2022-08-01T21:54:19.145448Z","shell.execute_reply.started":"2022-08-01T21:54:14.607688Z","shell.execute_reply":"2022-08-01T21:54:19.143976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:19.147679Z","iopub.execute_input":"2022-08-01T21:54:19.148052Z","iopub.status.idle":"2022-08-01T21:54:19.167506Z","shell.execute_reply.started":"2022-08-01T21:54:19.148019Z","shell.execute_reply":"2022-08-01T21:54:19.166259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom function","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:20.436288Z","iopub.execute_input":"2022-08-01T21:54:20.436779Z","iopub.status.idle":"2022-08-01T21:54:20.451962Z","shell.execute_reply.started":"2022-08-01T21:54:20.436731Z","shell.execute_reply":"2022-08-01T21:54:20.450411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate Amex metric for individual models","metadata":{}},{"cell_type":"code","source":"oofCols = [col for col in oof_df.columns if 'oof' in col]\n\nfor col in oofCols:\n    metric = amex_metric_mod(oof_df['target'], oof_df[col])\n    \n    print(f\"{col} : {metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:21.794001Z","iopub.execute_input":"2022-08-01T21:54:21.794409Z","iopub.status.idle":"2022-08-01T21:54:22.558712Z","shell.execute_reply.started":"2022-08-01T21:54:21.794375Z","shell.execute_reply":"2022-08-01T21:54:22.557391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple Averaging","metadata":{}},{"cell_type":"code","source":"oof_preds = []\nfor col in oofCols:\n    oof_preds.append(oof_df[col])\n\ny_avg = np.mean(np.array(oof_preds), axis=0)\nsa_metric = amex_metric_mod(oof_df['target'], y_avg)\n\nprint(f\"Simple Average: {sa_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:25.385420Z","iopub.execute_input":"2022-08-01T21:54:25.386783Z","iopub.status.idle":"2022-08-01T21:54:25.651849Z","shell.execute_reply.started":"2022-08-01T21:54:25.386732Z","shell.execute_reply":"2022-08-01T21:54:25.650835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted averaging","metadata":{}},{"cell_type":"code","source":"weights = [1,2,3]\n\ny_wtavg = np.zeros(len(oof_df))\n\nfor wt, col in zip(weights, oofCols):\n    y_wtavg = y_wtavg + (wt*oof_df[col])\n\ny_wtavg = y_wtavg / sum(weights)\n\nwa_metric = amex_metric_mod(oof_df['target'], y_wtavg)\n\nprint(f\"Weighted Average: {wa_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:27.283959Z","iopub.execute_input":"2022-08-01T21:54:27.285019Z","iopub.status.idle":"2022-08-01T21:54:27.559409Z","shell.execute_reply.started":"2022-08-01T21:54:27.284973Z","shell.execute_reply":"2022-08-01T21:54:27.558161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Rank averaging","metadata":{}},{"cell_type":"code","source":"rankPreds = []\n\nfor i, col in enumerate(oofCols):\n    globals()[f'y_rank{i+1}'] = oof_df[col].rank().values\n    rankPreds.append(globals()[f'y_rank{i+1}'])\n\ny_rankavg = np.mean(np.array(rankPreds), axis=0)\nra_metric = amex_metric_mod(oof_df['target'], y_rankavg)\n\nprint(f\"Rank Average: {ra_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:29.183884Z","iopub.execute_input":"2022-08-01T21:54:29.184299Z","iopub.status.idle":"2022-08-01T21:54:29.752948Z","shell.execute_reply.started":"2022-08-01T21:54:29.184265Z","shell.execute_reply":"2022-08-01T21:54:29.751601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted Rank Averaging","metadata":{}},{"cell_type":"code","source":"weights = [1,2,3]\n\ny_wtrankavg = np.zeros(len(oof_df))\n\nfor wt, pred in zip(weights, rankPreds):\n    y_wtrankavg = y_wtrankavg + (wt*pred)\n\ny_wtrankavg = y_wtrankavg / sum(weights)\n\nwra_metric = amex_metric_mod(oof_df['target'], y_wtrankavg)\n\nprint(f\"Weighted Rank Average: {wra_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:30.974191Z","iopub.execute_input":"2022-08-01T21:54:30.975434Z","iopub.status.idle":"2022-08-01T21:54:31.251679Z","shell.execute_reply.started":"2022-08-01T21:54:30.975386Z","shell.execute_reply":"2022-08-01T21:54:31.250288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hill climbing","metadata":{}},{"cell_type":"code","source":"RES = 20\nPATIENCE = 2\nDUPLICATES = True\n\n\npreds = oof_df[oofCols[-1]]\n\nprint(f\"Starting with {oofCols[-1]}\")\nhc_metric = amex_metric_mod(oof_df['target'], preds)\nprint(f\"Metric = {hc_metric}\")\n\nwhile True:\n    \n    if PATIENCE == 0:\n        break\n    \n    reduce = 1\n    \n    champ = amex_metric_mod(oof_df['target'], preds)\n    wtadd = 0\n    newcol = ''\n\n    for col in tqdm(oofCols):\n        for wt in range(RES):\n            temp = (wt/RES) * preds + (1-wt/RES) * oof_df[col]\n            chall = amex_metric_mod(oof_df['target'], temp)\n            \n            if chall > champ:\n                champ = chall\n                wtadd = (1-wt/RES)\n                newcol = col\n                newpred = temp\n                reduce = 0\n    print(f\"Adding {newcol} with weight {wtadd}\")\n    preds = newpred\n    hc_metric = amex_metric_mod(oof_df['target'], preds)\n    print(f\"Metric = {hc_metric}\")\n    \n    if reduce==1:\n        print(\"Metric has not improved\")\n        PATIENCE = PATIENCE - reduce","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:54:33.624412Z","iopub.execute_input":"2022-08-01T21:54:33.624852Z","iopub.status.idle":"2022-08-01T21:55:52.857393Z","shell.execute_reply.started":"2022-08-01T21:54:33.624814Z","shell.execute_reply":"2022-08-01T21:55:52.855960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Optimization","metadata":{}},{"cell_type":"code","source":"class OptimizeAmex:\n    def __init__(self):\n        self.coef_ = 0\n    \n    def _amex(self, coef, X, y):\n        x_coef = X * coef\n        predictions = np.sum(x_coef, axis=1)\n        amex_score = amex_metric_mod(y, predictions)\n        return -1.0 * amex_score\n    \n    def fit(self, X, y):\n        partial_loss = partial(self._amex, X=X, y=y)\n        init_coef = np.random.dirichlet(np.ones(X.shape[1]))\n        self.coef_ = fmin(partial_loss, init_coef, disp=False)\n    \n    def predict(self, X):\n        x_coef = X * self.coef_\n        predictions = np.sum(x_coef, axis=1)\n        return predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:56:35.128907Z","iopub.execute_input":"2022-08-01T21:56:35.129754Z","iopub.status.idle":"2022-08-01T21:56:35.140626Z","shell.execute_reply.started":"2022-08-01T21:56:35.129695Z","shell.execute_reply":"2022-08-01T21:56:35.139122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom functools import partial\nfrom scipy.optimize import fmin\n\ncoefs = []\n\nskf = KFold(n_splits=5, shuffle=True, random_state=42)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(oof_df, oof_df.target)):\n    \n    X_train = oof_df.loc[train_idx, oofCols]\n    y_train = oof_df.loc[train_idx, 'target']\n    \n    X_valid = oof_df.loc[valid_idx, oofCols]\n    y_valid = oof_df.loc[valid_idx, 'target']\n    \n    opt = OptimizeAmex()\n    opt.fit(X_train, y_train)\n    preds = opt.predict(X_valid)\n    score = amex_metric_mod(y_valid, preds)\n    \n    coefs.append(opt.coef_)\n    \n    print(f\"Fold {fold+1} score = {score}\")\n    print(opt.coef_)\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:56:36.963997Z","iopub.execute_input":"2022-08-01T21:56:36.964773Z","iopub.status.idle":"2022-08-01T21:57:49.241034Z","shell.execute_reply.started":"2022-08-01T21:56:36.964722Z","shell.execute_reply":"2022-08-01T21:57:49.239671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optcoefs = np.mean(coefs, axis=0)\n\npreds = 0\nfor wt, col in zip(optcoefs, oofCols):\n    preds += wt * oof_df[col]\n\nwopt_metric = amex_metric_mod(oof_df['target'], preds)\n\nprint(f\"Optimized weighted averaging: {wopt_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:57:49.243442Z","iopub.execute_input":"2022-08-01T21:57:49.244489Z","iopub.status.idle":"2022-08-01T21:57:49.501272Z","shell.execute_reply.started":"2022-08-01T21:57:49.244436Z","shell.execute_reply":"2022-08-01T21:57:49.499536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison","metadata":{}},{"cell_type":"code","source":"print(f\"Simple Average: {sa_metric}\")\nprint(f\"Weighted Average: {wa_metric}\")\nprint(f\"Rank Average: {ra_metric}\")\nprint(f\"Weighted Rank Average: {wra_metric}\")\nprint(f\"Hill Climbing = {hc_metric}\")\nprint(f\"Optimized weighted averaging: {wopt_metric}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T21:58:47.775514Z","iopub.execute_input":"2022-08-01T21:58:47.776049Z","iopub.status.idle":"2022-08-01T21:58:47.785585Z","shell.execute_reply.started":"2022-08-01T21:58:47.776009Z","shell.execute_reply":"2022-08-01T21:58:47.783701Z"},"trusted":true},"execution_count":null,"outputs":[]}]}