{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ensemble fork with weights\n\n## I forked this notebook: https://www.kaggle.com/code/apoorvbhardwaj/amex-early-ensemble/notebook?scriptVersionId=97132599\n## Please give that notebook an upvote if you give this one an upvote...\n\nI took the above notebook and played around with model weights. Pretty simple idea added to the above simple idea.\n\nHere's a good explanation of the idea of doing a weighted ensemble.\nhttps://machinelearningmastery.com/weighted-average-ensemble-with-python/\n\n\nAnyway, just a slight improvement over the origial linked notebook above\n* Original score = 0.794\n* This score = 0.794 ++ (the leaderboard needs to add some decimal places...)","metadata":{"papermill":{"duration":0.008894,"end_time":"2021-12-31T14:48:41.466849","exception":false,"start_time":"2021-12-31T14:48:41.457955","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport plotly.express as px","metadata":{"papermill":{"duration":2.310801,"end_time":"2021-12-31T14:48:43.785659","exception":false,"start_time":"2021-12-31T14:48:41.474858","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:10.609545Z","iopub.execute_input":"2022-06-01T03:42:10.609805Z","iopub.status.idle":"2022-06-01T03:42:10.623976Z","shell.execute_reply.started":"2022-06-01T03:42:10.609772Z","shell.execute_reply":"2022-06-01T03:42:10.623243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targetName = 'prediciton'\ncompetitionDir = '../input/amex-default-prediction'\nsubmission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"papermill":{"duration":0.343502,"end_time":"2021-12-31T14:48:44.137086","exception":false,"start_time":"2021-12-31T14:48:43.793584","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:10.625308Z","iopub.execute_input":"2022-06-01T03:42:10.625815Z","iopub.status.idle":"2022-06-01T03:42:11.870208Z","shell.execute_reply.started":"2022-06-01T03:42:10.625774Z","shell.execute_reply":"2022-06-01T03:42:11.869357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nregex = \"([^\\\\s]+(\\\\.(?i)(csv))$)\"\np = re.compile(regex)","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.871482Z","iopub.execute_input":"2022-06-01T03:42:11.871726Z","iopub.status.idle":"2022-06-01T03:42:11.876567Z","shell.execute_reply.started":"2022-06-01T03:42:11.871696Z","shell.execute_reply":"2022-06-01T03:42:11.875537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []","metadata":{"papermill":{"duration":17.542154,"end_time":"2021-12-31T14:49:01.701331","exception":false,"start_time":"2021-12-31T14:48:44.159177","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-01T03:42:11.877773Z","iopub.execute_input":"2022-06-01T03:42:11.878045Z","iopub.status.idle":"2022-06-01T03:42:11.894216Z","shell.execute_reply.started":"2022-06-01T03:42:11.878013Z","shell.execute_reply":"2022-06-01T03:42:11.893091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.89779Z","iopub.execute_input":"2022-06-01T03:42:11.898351Z","iopub.status.idle":"2022-06-01T03:42:11.908351Z","shell.execute_reply.started":"2022-06-01T03:42:11.898294Z","shell.execute_reply":"2022-06-01T03:42:11.907301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted average\nAs per the little function below, I wrote a little formula for weighting the predictions that is based on how well the model it came from scored. The best model is weighted 1, and the worst is weighted somewhere around 0.05.\n\nThe article that's linked below has some more elegant ways to do the weighting if you actually buld the models youself and can therefore do some cross validation, but here, since I'm just pulling the output from other people's hard work, I can't and can just do the crude little thing below. \nhttps://machinelearningmastery.com/weighted-average-ensemble-with-python/","metadata":{}},{"cell_type":"code","source":"#crude formula for model weighting.\n# Using this formaula the best model's weight = 1, the worst model's weight ~0.05\ndef model_weight(model_loss, worst_loss, best_loss):\n    return 1- ((best_loss - model_loss)/(best_loss-(worst_loss - 0.01)))","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:42:11.909627Z","iopub.execute_input":"2022-06-01T03:42:11.910559Z","iopub.status.idle":"2022-06-01T03:42:11.924393Z","shell.execute_reply.started":"2022-06-01T03:42:11.91051Z","shell.execute_reply":"2022-06-01T03:42:11.923026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best = 0.794\nworst = 0.783\nw1 = model_weight(.79, worst, best)\nw2 = model_weight(0.794, worst, best) \nw4 = model_weight(0.791, worst, best)\nw6 = model_weight(0.794, worst, best)\nw7 = model_weight(0.783, worst, best)\n#w8 = model_weight(.715, worst, best)\n\n\ndf = pd.read_csv('../input/amex-lightgbm-quickstart/submission.csv')\ndf['prediction'] = df['prediction']*w1\npreds.append((df['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df2 = pd.read_csv('../input/amex-catboost-0-793/submission_cat_0.7887909351813577.csv')\ndf2 = pd.read_csv('../input/amex-catboost-rounding-trick/submission_cat_0.7923765971248049.csv')\ndf2['prediction'] = df2['prediction']*w2\npreds.append((df2['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df3 = pd.read_csv('../input/fork-of-amex-lightgbm-quickstart/submission.csv')\n#preds.append((df3['prediction']))\n#print(np.column_stack(preds).shape)\n\ndf4 = pd.read_csv('../input/lb-split-downsampling-trick-amex/submission.csv')\ndf4['prediction'] = df4['prediction']*w4\nprint(np.column_stack(preds).shape)\npreds.append(df4['prediction'])\n\n#df5 = pd.read_csv('../input/amex-lgbm-features-eng/submission.csv')\n#preds.append((df5['prediction']))\ndf6 = pd.read_csv('../input/amex-lightautoml-starter/lightautoml_tabularautoml.csv')\ndf6['prediction'] = df6['prediction']*w6\npreds.append(df6['prediction'])\nprint(np.column_stack(preds).shape)\n\ndf7 = pd.read_csv('../input/amex-default-prediction-keras-starter/my_submission.csv')\ndf7['prediction'] = df7['prediction']*w7\npreds.append(df7['prediction'])\nprint(np.column_stack(preds).shape)\n\n\n# df8 = pd.read_csv('../input/best-correlated-features-with-low-correlation/submission.csv')\n# df8['prediction'] = df8['prediction']*w8\n# preds.append(df8['prediction'])\n# print(np.column_stack(preds).shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:21.479966Z","iopub.execute_input":"2022-06-01T03:35:21.480265Z","iopub.status.idle":"2022-06-01T03:35:33.846101Z","shell.execute_reply.started":"2022-06-01T03:35:21.480235Z","shell.execute_reply":"2022-06-01T03:35:33.84515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.848388Z","iopub.execute_input":"2022-06-01T03:35:33.848962Z","iopub.status.idle":"2022-06-01T03:35:33.86379Z","shell.execute_reply.started":"2022-06-01T03:35:33.848885Z","shell.execute_reply":"2022-06-01T03:35:33.862776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nblend_ss = submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.865032Z","iopub.execute_input":"2022-06-01T03:35:33.865254Z","iopub.status.idle":"2022-06-01T03:35:33.893804Z","shell.execute_reply.started":"2022-06-01T03:35:33.865227Z","shell.execute_reply":"2022-06-01T03:35:33.89288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss['prediction'] = (np.sum(np.column_stack(preds), axis=1) / (w1+w2+w4+w6+w7))\nblend_ss","metadata":{"execution":{"iopub.status.busy":"2022-06-01T03:35:33.89592Z","iopub.execute_input":"2022-06-01T03:35:33.896433Z","iopub.status.idle":"2022-06-01T03:35:33.985908Z","shell.execute_reply.started":"2022-06-01T03:35:33.896377Z","shell.execute_reply":"2022-06-01T03:35:33.985304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T21:35:48.044958Z","iopub.status.idle":"2022-05-31T21:35:48.045258Z","shell.execute_reply.started":"2022-05-31T21:35:48.045093Z","shell.execute_reply":"2022-05-31T21:35:48.045109Z"},"trusted":true},"execution_count":null,"outputs":[]}]}