{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ensemble fork with weights - Fork of the fork of the fork...\n\n## I forked this notebook:\nhttps://www.kaggle.com/code/beezus666/ensemble-weighted-average\n\nhttps://www.kaggle.com/code/apoorvbhardwaj/amex-early-ensemble/notebook?scriptVersionId=97132599\n\n## Please give that notebook an upvote if you give this one an upvote...\n","metadata":{"papermill":{"duration":0.008894,"end_time":"2021-12-31T14:48:41.466849","exception":false,"start_time":"2021-12-31T14:48:41.457955","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport plotly.express as px","metadata":{"papermill":{"duration":2.310801,"end_time":"2021-12-31T14:48:43.785659","exception":false,"start_time":"2021-12-31T14:48:41.474858","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:10.609545Z","iopub.execute_input":"2022-06-01T03:42:10.609805Z","iopub.status.idle":"2022-06-01T03:42:10.623976Z","shell.execute_reply.started":"2022-06-01T03:42:10.609772Z","shell.execute_reply":"2022-06-01T03:42:10.623243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targetName = 'prediciton'\ncompetitionDir = '../input/amex-default-prediction'\nsubmission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"papermill":{"duration":0.343502,"end_time":"2021-12-31T14:48:44.137086","exception":false,"start_time":"2021-12-31T14:48:43.793584","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:10.625308Z","iopub.execute_input":"2022-06-01T03:42:10.625815Z","iopub.status.idle":"2022-06-01T03:42:11.870208Z","shell.execute_reply.started":"2022-06-01T03:42:10.625774Z","shell.execute_reply":"2022-06-01T03:42:11.869357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nregex = \"([^\\\\s]+(\\\\.(?i)(csv))$)\"\np = re.compile(regex)","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.871482Z","iopub.execute_input":"2022-06-01T03:42:11.871726Z","iopub.status.idle":"2022-06-01T03:42:11.876567Z","shell.execute_reply.started":"2022-06-01T03:42:11.871696Z","shell.execute_reply":"2022-06-01T03:42:11.875537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []","metadata":{"papermill":{"duration":17.542154,"end_time":"2021-12-31T14:49:01.701331","exception":false,"start_time":"2021-12-31T14:48:44.159177","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:11.877773Z","iopub.execute_input":"2022-06-01T03:42:11.878045Z","iopub.status.idle":"2022-06-01T03:42:11.894216Z","shell.execute_reply.started":"2022-06-01T03:42:11.878013Z","shell.execute_reply":"2022-06-01T03:42:11.893091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.89779Z","iopub.execute_input":"2022-06-01T03:42:11.898351Z","iopub.status.idle":"2022-06-01T03:42:11.908351Z","shell.execute_reply.started":"2022-06-01T03:42:11.898294Z","shell.execute_reply":"2022-06-01T03:42:11.907301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted average\nAs per the little function below, I wrote a little formula for weighting the predictions that is based on how well the model it came from scored. The best model is weighted 1, and the worst is weighted somewhere around 0.05.\n\nThe article that's linked below has some more elegant ways to do the weighting if you actually buld the models youself and can therefore do some cross validation, but here, since I'm just pulling the output from other people's hard work, I can't and can just do the crude little thing below. \nhttps://machinelearningmastery.com/weighted-average-ensemble-with-python/","metadata":{}},{"cell_type":"code","source":"#crude formula for model weighting.\n# Using this formaula the best model's weight = 1, the worst model's weight ~0.05\ndef model_weight(model_loss, worst_loss, best_loss):\n    return 1- ((best_loss - model_loss)/(best_loss-(worst_loss - 0.01)))","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.909627Z","iopub.execute_input":"2022-06-01T03:42:11.910559Z","iopub.status.idle":"2022-06-01T03:42:11.924393Z","shell.execute_reply.started":"2022-06-01T03:42:11.91051Z","shell.execute_reply":"2022-06-01T03:42:11.923026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best = 0.796\nworst = 0.783\nw1 = model_weight(.79, worst, best)\nw2 = model_weight(0.794, worst, best) \nw4 = model_weight(0.791, worst, best)\nw6 = model_weight(0.794, worst, best)\nw7 = model_weight(0.783, worst, best)\nw8 = model_weight(0.796, worst, best)\n\nprint(w1,w2,w4,w6,w7,w8)\n\n\ndf = pd.read_csv('../input/amex-lightgbm-quickstart/submission.csv')\ndf['prediction'] = df['prediction']*w1\npreds.append((df['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df2 = pd.read_csv('../input/amex-catboost-0-793/submission_cat_0.7887909351813577.csv')\ndf2 = pd.read_csv('../input/amex-catboost-rounding-trick/submission_cat_0.7923765971248049.csv')\ndf2['prediction'] = df2['prediction']*w2\npreds.append((df2['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df3 = pd.read_csv('../input/fork-of-amex-lightgbm-quickstart/submission.csv')\n#preds.append((df3['prediction']))\n#print(np.column_stack(preds).shape)\n\ndf4 = pd.read_csv('../input/lb-split-downsampling-trick-amex/submission.csv')\ndf4['prediction'] = df4['prediction']*w4\nprint(np.column_stack(preds).shape)\npreds.append(df4['prediction'])\n\n#df5 = pd.read_csv('../input/amex-lgbm-features-eng/submission.csv')\n#preds.append((df5['prediction']))\ndf6 = pd.read_csv('../input/amex-lightautoml-starter/lightautoml_tabularautoml.csv')\ndf6['prediction'] = df6['prediction']*w6\npreds.append(df6['prediction'])\nprint(np.column_stack(preds).shape)\n\ndf7 = pd.read_csv('../input/amex-default-prediction-keras-starter/my_submission.csv')\ndf7['prediction'] = df7['prediction']*w7\npreds.append(df7['prediction'])\nprint(np.column_stack(preds).shape)\n\n\ndf8 = pd.read_csv('../input/amex-lgbm-dart-cv-0-7963/test_lgbm_baseline_5fold_seed42.csv')\ndf8['prediction'] = df8['prediction']*w8\npreds.append(df8['prediction'])\nprint(np.column_stack(preds).shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:21.479966Z","iopub.execute_input":"2022-06-01T03:35:21.480265Z","iopub.status.idle":"2022-06-01T03:35:33.846101Z","shell.execute_reply.started":"2022-06-01T03:35:21.480235Z","shell.execute_reply":"2022-06-01T03:35:33.84515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.848388Z","iopub.execute_input":"2022-06-01T03:35:33.848962Z","iopub.status.idle":"2022-06-01T03:35:33.86379Z","shell.execute_reply.started":"2022-06-01T03:35:33.848885Z","shell.execute_reply":"2022-06-01T03:35:33.862776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nblend_ss = submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.865032Z","iopub.execute_input":"2022-06-01T03:35:33.865254Z","iopub.status.idle":"2022-06-01T03:35:33.893804Z","shell.execute_reply.started":"2022-06-01T03:35:33.865227Z","shell.execute_reply":"2022-06-01T03:35:33.89288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss['prediction'] = (np.sum(np.column_stack(preds), axis=1) / (w1+w2+w4+w6+w7+w8))\nblend_ss","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.89592Z","iopub.execute_input":"2022-06-01T03:35:33.896433Z","iopub.status.idle":"2022-06-01T03:35:33.985908Z","shell.execute_reply.started":"2022-06-01T03:35:33.896377Z","shell.execute_reply":"2022-06-01T03:35:33.985304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T21:35:48.044958Z","iopub.status.idle":"2022-05-31T21:35:48.045258Z","shell.execute_reply.started":"2022-05-31T21:35:48.045093Z","shell.execute_reply":"2022-05-31T21:35:48.045109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Median\n\nSubmission score: 0.796","metadata":{}},{"cell_type":"code","source":"%%time\nblend_2 = submission.copy()\nblend_2['prediction'] = (np.median(np.column_stack(preds), axis=1))\nblend_2.to_csv('submission_median.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple average\n\nScore equal to the weighted model: 0.797","metadata":{}},{"cell_type":"code","source":"\n%%time\nblend_3 = submission.copy()\nblend_3['prediction'] = (np.mean(np.column_stack(preds), axis=1))\nblend_3.to_csv('submission_mean.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}