{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Exponential weighted ensemble**","metadata":{}},{"cell_type":"markdown","source":"This notebook is an extension of https://www.kaggle.com/code/hikarutabata/exponential-ensemble. \n\nIn this notebook, I propose an alternative to the calibration done by @hikarutabata, which consists of weightning each prediction by its probability density, simple isn't ?\n\nThis weightning strategy gives the same results as https://www.kaggle.com/code/hikarutabata/exponential-ensemble.","metadata":{}},{"cell_type":"markdown","source":"**Notebooks used**","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/swimmy/tuffline-amex-anotherfeaturelgbm\n\nhttps://www.kaggle.com/code/adhithyasrinivasan/amex-feature-ensemble\n\nhttps://www.kaggle.com/code/zb1373/blend-boosting-study\n\nhttps://www.kaggle.com/code/manavtrivedi/tuffline-plotly-amex","metadata":{}},{"cell_type":"markdown","source":"**ensemble**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob","metadata":{"execution":{"iopub.status.busy":"2022-08-24T12:01:29.541520Z","iopub.execute_input":"2022-08-24T12:01:29.541910Z","iopub.status.idle":"2022-08-24T12:01:29.547893Z","shell.execute_reply.started":"2022-08-24T12:01:29.541880Z","shell.execute_reply":"2022-08-24T12:01:29.546223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\ndisplay(sub)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T12:09:40.155436Z","iopub.execute_input":"2022-08-24T12:09:40.155853Z","iopub.status.idle":"2022-08-24T12:09:41.391112Z","shell.execute_reply.started":"2022-08-24T12:09:40.155812Z","shell.execute_reply":"2022-08-24T12:09:41.389947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = [\n    \"../input/tuffline-plotly-amex/submission.csv\",\n    \"../input/blend-boosting-study/submission.csv\",\n    \"../input/amex-feature-ensemble/submission.csv\",\n    \"../input/tuffline-amex-anotherfeaturelgbm/submission.csv\"\n]","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:02:31.938017Z","iopub.execute_input":"2022-08-24T13:02:31.938841Z","iopub.status.idle":"2022-08-24T13:02:31.943990Z","shell.execute_reply.started":"2022-08-24T13:02:31.938801Z","shell.execute_reply":"2022-08-24T13:02:31.942847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nl = []\n\nfor df in dfs:\n    l.append(pd.read_csv(df)[\"prediction\"])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:03:34.837067Z","iopub.execute_input":"2022-08-24T13:03:34.837505Z","iopub.status.idle":"2022-08-24T13:03:41.456367Z","shell.execute_reply.started":"2022-08-24T13:03:34.837469Z","shell.execute_reply":"2022-08-24T13:03:41.455070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see here, we observe the probability distribution of each model's prediction. Our goal is to give more weight to model's prediction sample if it's probability density is higher  ","metadata":{}},{"cell_type":"code","source":"fig,ax  = plt.subplots(1,figsize=(20,5))\nsns.distplot(l[0],ax=ax)\nsns.distplot(l[1],ax=ax)\nsns.distplot(l[2],ax=ax)\nsns.distplot(l[3],ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:14:42.704050Z","iopub.execute_input":"2022-08-24T13:14:42.704641Z","iopub.status.idle":"2022-08-24T13:14:58.096333Z","shell.execute_reply.started":"2022-08-24T13:14:42.704573Z","shell.execute_reply":"2022-08-24T13:14:58.095057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How are we going to do this ?\n\nSimple, we extracts the probability density using `norm.cdf` by specifiying each latent variables of each distribution","metadata":{}},{"cell_type":"code","source":"from scipy.stats import norm\n\n\nprobability_density_0  = norm.cdf(l[0], np.mean(l[0]), np.std(l[0]))\nprobability_density_1  = norm.cdf(l[1], np.mean(l[1]), np.std(l[1]))\n#probability_density_2  = norm.cdf(l[2], np.mean(l[2]), np.std(l[2]))\n#probability_density_3  = norm.cdf(l[3], np.mean(l[3]), np.std(l[3]))\n\nsum_density = probability_density_0+ probability_density_1 #+probability_density_2 #+ probability_density_3","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:04:49.920139Z","iopub.execute_input":"2022-08-24T13:04:49.920572Z","iopub.status.idle":"2022-08-24T13:04:50.153431Z","shell.execute_reply.started":"2022-08-24T13:04:49.920538Z","shell.execute_reply":"2022-08-24T13:04:50.152431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, we simply do an average over the probabilities density that represent the corresponding weights :)","metadata":{}},{"cell_type":"code","source":"sub['prediction'] = (probability_density_0*l[0] + probability_density_1*l[1]  )  / sum_density ","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:05:43.722713Z","iopub.execute_input":"2022-08-24T13:05:43.723176Z","iopub.status.idle":"2022-08-24T13:05:43.750819Z","shell.execute_reply.started":"2022-08-24T13:05:43.723137Z","shell.execute_reply":"2022-08-24T13:05:43.749680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this figure, we can observe the distribution of the final prediction.","metadata":{}},{"cell_type":"code","source":"fig,ax  = plt.subplots(1,figsize=(20,5))\nsns.distplot(sub['prediction'],ax=ax)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:05:45.304363Z","iopub.execute_input":"2022-08-24T13:05:45.305494Z","iopub.status.idle":"2022-08-24T13:05:49.368861Z","shell.execute_reply.started":"2022-08-24T13:05:45.305446Z","shell.execute_reply":"2022-08-24T13:05:49.367486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T13:05:54.059993Z","iopub.execute_input":"2022-08-24T13:05:54.060704Z","iopub.status.idle":"2022-08-24T13:05:57.287940Z","shell.execute_reply.started":"2022-08-24T13:05:54.060659Z","shell.execute_reply":"2022-08-24T13:05:57.286634Z"},"trusted":true},"execution_count":null,"outputs":[]}]}