{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-05T19:20:42.178987Z","iopub.execute_input":"2022-06-05T19:20:42.180044Z","iopub.status.idle":"2022-06-05T19:20:42.273006Z","shell.execute_reply.started":"2022-06-05T19:20:42.180004Z","shell.execute_reply":"2022-06-05T19:20:42.272064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:21:01.304261Z","iopub.execute_input":"2022-06-05T19:21:01.304669Z","iopub.status.idle":"2022-06-05T19:21:02.836424Z","shell.execute_reply.started":"2022-06-05T19:21:01.304636Z","shell.execute_reply":"2022-06-05T19:21:02.835543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targetName = 'prediciton'\ncompetitionDir = '../input/amex-default-prediction'\nsubmission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:21:11.257498Z","iopub.execute_input":"2022-06-05T19:21:11.258203Z","iopub.status.idle":"2022-06-05T19:21:13.114333Z","shell.execute_reply.started":"2022-06-05T19:21:11.258165Z","shell.execute_reply":"2022-06-05T19:21:13.113294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nregex = \"([^\\\\s]+(\\\\.(?i)(csv))$)\"\np = re.compile(regex)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:21:21.521453Z","iopub.execute_input":"2022-06-05T19:21:21.521823Z","iopub.status.idle":"2022-06-05T19:21:21.527172Z","shell.execute_reply.started":"2022-06-05T19:21:21.521793Z","shell.execute_reply":"2022-06-05T19:21:21.525859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:38:02.598938Z","iopub.execute_input":"2022-06-05T19:38:02.599314Z","iopub.status.idle":"2022-06-05T19:38:02.603033Z","shell.execute_reply.started":"2022-06-05T19:38:02.599283Z","shell.execute_reply":"2022-06-05T19:38:02.602118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:21:41.713991Z","iopub.execute_input":"2022-06-05T19:21:41.714387Z","iopub.status.idle":"2022-06-05T19:21:41.780514Z","shell.execute_reply.started":"2022-06-05T19:21:41.714354Z","shell.execute_reply":"2022-06-05T19:21:41.779615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#crude formula for model weighting.\n# Using this formaula the best model's weight = 1, the worst model's weight ~0.05\ndef model_weight(model_loss, worst_loss, best_loss):\n    return 1- ((best_loss - model_loss)/(best_loss-(worst_loss - 0.01)))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:21:57.517815Z","iopub.execute_input":"2022-06-05T19:21:57.518175Z","iopub.status.idle":"2022-06-05T19:21:57.522932Z","shell.execute_reply.started":"2022-06-05T19:21:57.518145Z","shell.execute_reply":"2022-06-05T19:21:57.521837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best = 0.794\nworst = 0.783\nw1 = model_weight(.79, worst, best)\nw2 = model_weight(0.794, worst, best) \nw4 = model_weight(0.791, worst, best)\nw6 = model_weight(0.794, worst, best)\nw7 = model_weight(0.783, worst, best)\n# w8 = model_weight(0.796, worst, best)\n\n\ndf = pd.read_csv('../input/amex-lightgbm-quickstart/submission.csv')\ndf['prediction'] = df['prediction']*w1\npreds.append((df['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df2 = pd.read_csv('../input/amex-catboost-0-793/submission_cat_0.7887909351813577.csv')\ndf2 = pd.read_csv('../input/amex-catboost-rounding-trick/submission_cat_0.7923765971248049.csv')\ndf2['prediction'] = df2['prediction']*w2\npreds.append((df2['prediction']))\nprint(np.column_stack(preds).shape)\n\n#df3 = pd.read_csv('../input/fork-of-amex-lightgbm-quickstart/submission.csv')\n#preds.append((df3['prediction']))\n#print(np.column_stack(preds).shape)\n\ndf4 = pd.read_csv('../input/lb-split-downsampling-trick-amex/submission.csv')\ndf4['prediction'] = df4['prediction']*w4\nprint(np.column_stack(preds).shape)\npreds.append(df4['prediction'])\n\n#df5 = pd.read_csv('../input/amex-lgbm-features-eng/submission.csv')\n#preds.append((df5['prediction']))\ndf6 = pd.read_csv('../input/amex-lightautoml-starter/lightautoml_tabularautoml.csv')\ndf6['prediction'] = df6['prediction']*w6\npreds.append(df6['prediction'])\nprint(np.column_stack(preds).shape)\n\ndf7 = pd.read_csv('../input/amex-default-prediction-keras-starter/my_submission.csv')\ndf7['prediction'] = df7['prediction']*w7\npreds.append(df7['prediction'])\nprint(np.column_stack(preds).shape)\n\n\n# df8 = pd.read_csv('../input/ensemble-weighted-average/submission.csv')\n# df8['prediction'] = df8['prediction']*w8\n# preds.append(df8['prediction'])\n# print(np.column_stack(preds).shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:38:06.983601Z","iopub.execute_input":"2022-06-05T19:38:06.984122Z","iopub.status.idle":"2022-06-05T19:38:13.054285Z","shell.execute_reply.started":"2022-06-05T19:38:06.984084Z","shell.execute_reply":"2022-06-05T19:38:13.053321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:38:13.056572Z","iopub.execute_input":"2022-06-05T19:38:13.057038Z","iopub.status.idle":"2022-06-05T19:38:13.068420Z","shell.execute_reply.started":"2022-06-05T19:38:13.056993Z","shell.execute_reply":"2022-06-05T19:38:13.067485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss = submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:41:08.623283Z","iopub.execute_input":"2022-06-05T19:41:08.623644Z","iopub.status.idle":"2022-06-05T19:41:08.641509Z","shell.execute_reply.started":"2022-06-05T19:41:08.623615Z","shell.execute_reply":"2022-06-05T19:41:08.640387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss['prediction'] = (np.sum(np.column_stack(preds), axis=1) / (w1+w2+w4+w6+w7))\nblend_ss","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:41:35.757453Z","iopub.execute_input":"2022-06-05T19:41:35.757816Z","iopub.status.idle":"2022-06-05T19:41:35.822364Z","shell.execute_reply.started":"2022-06-05T19:41:35.757788Z","shell.execute_reply":"2022-06-05T19:41:35.821467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_ss.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:41:38.198147Z","iopub.execute_input":"2022-06-05T19:41:38.198660Z","iopub.status.idle":"2022-06-05T19:41:41.335432Z","shell.execute_reply.started":"2022-06-05T19:41:38.198621Z","shell.execute_reply":"2022-06-05T19:41:41.334669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T19:38:44.114199Z","iopub.execute_input":"2022-06-05T19:38:44.114595Z","iopub.status.idle":"2022-06-05T19:38:44.125320Z","shell.execute_reply.started":"2022-06-05T19:38:44.114565Z","shell.execute_reply":"2022-06-05T19:38:44.123946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}