{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-11T18:36:40.810429Z","iopub.execute_input":"2022-04-11T18:36:40.81092Z","iopub.status.idle":"2022-04-11T18:36:40.843123Z","shell.execute_reply.started":"2022-04-11T18:36:40.81079Z","shell.execute_reply":"2022-04-11T18:36:40.84169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Amex Metric\n\nThis is a python version of the metric for the Amex competition. Additional details can be found on the competition [Evaluation page](https://www.kaggle.com/competitions/amex-default-prediction/overview/evaluation).","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:48:30.894058Z","iopub.execute_input":"2022-04-11T18:48:30.894376Z","iopub.status.idle":"2022-04-11T18:48:30.912501Z","shell.execute_reply.started":"2022-04-11T18:48:30.894327Z","shell.execute_reply":"2022-04-11T18:48:30.911441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple Benchmark\n\nWe can create a simple benchark using the average of the feature `P_2` for each customer.","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\n    input_path / 'train_data.csv',\n    index_col='customer_ID',\n    usecols=['customer_ID', 'P_2'])\n\ntrain_labels = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:43:47.939939Z","iopub.execute_input":"2022-04-11T18:43:47.940251Z","iopub.status.idle":"2022-04-11T18:46:14.158219Z","shell.execute_reply.started":"2022-04-11T18:43:47.94022Z","shell.execute_reply":"2022-04-11T18:46:14.155702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ave_p2 = (train_data\n          .groupby('customer_ID')\n          .mean()\n          .rename(columns={'P_2': 'prediction'}))\n\n# Scale the mean P_2 by the max value and take the compliment\nave_p2['prediction'] = 1.0 - (ave_p2['prediction'] / ave_p2['prediction'].max())","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:49:20.015083Z","iopub.execute_input":"2022-04-11T18:49:20.015424Z","iopub.status.idle":"2022-04-11T18:49:21.617247Z","shell.execute_reply.started":"2022-04-11T18:49:20.01539Z","shell.execute_reply":"2022-04-11T18:49:21.616104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric(train_labels, ave_p2)) # 0.572773","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:49:26.344222Z","iopub.execute_input":"2022-04-11T18:49:26.344605Z","iopub.status.idle":"2022-04-11T18:49:27.996226Z","shell.execute_reply.started":"2022-04-11T18:49:26.34457Z","shell.execute_reply":"2022-04-11T18:49:27.994963Z"},"trusted":true},"execution_count":null,"outputs":[]}]}