{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exponential Weighted Ensemble\nThis is an exponentially weighted ensemble example, inspired on this [Spaceship Titanic notebook](https://www.kaggle.com/code/jbomitchell/exponentially-weighted-ensemble-spaceship-titanic).\n\nThe forumla used is: exp(b*(x-S))\n\nx is the LB score of each model. Larger scores are better, hence the weights get larger as x increases and get smaller as x decreases (The better the model, the bigger the weight).\n\nb is the one meaningfully adjustable parameter of the model, the larger b is then the faster the weights decay as the score gets worse. S is a calibration parameter defined such that if S is set to the best single model score then the highest unnormalised weight exp(b*(x-S)) is 1.0, which is convenient, but not essential.\n\nq is the sum of the unnormalised weights. Once all weights have been calculated, these weights are normalised by dividing them all by q.\n\n","metadata":{}},{"cell_type":"code","source":"# Fancy libraries\nimport numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-15T15:29:34.270771Z","iopub.execute_input":"2022-08-15T15:29:34.271847Z","iopub.status.idle":"2022-08-15T15:29:34.277093Z","shell.execute_reply.started":"2022-08-15T15:29:34.271801Z","shell.execute_reply":"2022-08-15T15:29:34.275836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters for exponential ensemble\nb = 4000.0\nS = 0.7991\nq = 0.0","metadata":{"execution":{"iopub.status.busy":"2022-08-15T15:29:34.840967Z","iopub.execute_input":"2022-08-15T15:29:34.841377Z","iopub.status.idle":"2022-08-15T15:29:34.846798Z","shell.execute_reply.started":"2022-08-15T15:29:34.841345Z","shell.execute_reply":"2022-08-15T15:29:34.845794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# placeholder submission\nsubmission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n\nsub = submission.copy()\nsub.sort_values(by=['customer_ID'], inplace=True)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T15:29:36.027494Z","iopub.execute_input":"2022-08-15T15:29:36.028402Z","iopub.status.idle":"2022-08-15T15:29:37.650252Z","shell.execute_reply.started":"2022-08-15T15:29:36.028360Z","shell.execute_reply":"2022-08-15T15:29:37.649285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load fancy models from fellow kagglers\nsub1 = pd.read_csv('../input/amex-lgbm-dart-cv-0-7977/test_lgbm_baseline_5fold_seed_blend.csv')\nsub1.sort_values(by=['customer_ID'], inplace=True)\n\nsub2 = pd.read_csv('../input/amex-lgbm-dart-cv-0-7963-improved/submission.csv')\nsub2.sort_values(by=['customer_ID'], inplace=True)\n\nsub3 = pd.read_csv('../input/the-fine-art-of-hyperparameter-tuning/submission.csv')\nsub3.sort_values(by=['customer_ID'], inplace=True)\n\nsub4 = pd.read_csv('../input/amex-features-the-best-of-both-worlds/submission.csv')\nsub4.sort_values(by=['customer_ID'], inplace=True)\n\nsub5 = pd.read_csv('../input/overfitting-public-lb/submission.csv')\nsub5.sort_values(by=['customer_ID'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T15:29:37.651751Z","iopub.execute_input":"2022-08-15T15:29:37.652142Z","iopub.status.idle":"2022-08-15T15:29:46.820498Z","shell.execute_reply.started":"2022-08-15T15:29:37.652114Z","shell.execute_reply":"2022-08-15T15:29:46.819303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Avengers ensemble! \nsub['prediction'] = sub1['prediction']*np.exp(b*(0.79905-S))\nq = q + np.exp(b*(0.79905-S))\nsub['prediction'] = sub['prediction'] + sub2['prediction']*np.exp(b*(0.79909-S))\nq = q + np.exp(b*(0.79909-S))\nsub['prediction'] = sub['prediction'] + sub3['prediction']*np.exp(b*(0.79909-S))\nq = q + np.exp(b*(0.79909-S))\nsub['prediction'] = sub['prediction'] + sub4['prediction']*np.exp(b*(0.79905-S))\nq = q + np.exp(b*(0.79905-S))\nsub['prediction'] = sub['prediction'] + sub5['prediction']*np.exp(b*(0.7991-S))\nq = q + np.exp(b*(0.7991-S))\n\nsub['prediction'] = sub['prediction']/q\n\n\nprint(q)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T15:31:51.901956Z","iopub.execute_input":"2022-08-15T15:31:51.902457Z","iopub.status.idle":"2022-08-15T15:31:51.950803Z","shell.execute_reply.started":"2022-08-15T15:31:51.902399Z","shell.execute_reply":"2022-08-15T15:31:51.949856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}