{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-10T13:26:21.408065Z","iopub.execute_input":"2022-06-10T13:26:21.408391Z","iopub.status.idle":"2022-06-10T13:26:21.413661Z","shell.execute_reply.started":"2022-06-10T13:26:21.408333Z","shell.execute_reply":"2022-06-10T13:26:21.412398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We copied the python implementation from the [competition host's notebook](https://www.kaggle.com/code/inversion/amex-competition-metric-python)","metadata":{}},{"cell_type":"code","source":"def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T13:26:23.431766Z","iopub.execute_input":"2022-06-10T13:26:23.432279Z","iopub.status.idle":"2022-06-10T13:26:23.441667Z","shell.execute_reply.started":"2022-06-10T13:26:23.432229Z","shell.execute_reply":"2022-06-10T13:26:23.440765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T13:26:25.99003Z","iopub.execute_input":"2022-06-10T13:26:25.99034Z","iopub.status.idle":"2022-06-10T13:26:25.996366Z","shell.execute_reply.started":"2022-06-10T13:26:25.990308Z","shell.execute_reply":"2022-06-10T13:26:25.995197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T13:26:28.750389Z","iopub.execute_input":"2022-06-10T13:26:28.751184Z","iopub.status.idle":"2022-06-10T13:26:28.758603Z","shell.execute_reply.started":"2022-06-10T13:26:28.751146Z","shell.execute_reply":"2022-06-10T13:26:28.757579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T13:26:31.18289Z","iopub.execute_input":"2022-06-10T13:26:31.183202Z","iopub.status.idle":"2022-06-10T13:26:31.188214Z","shell.execute_reply.started":"2022-06-10T13:26:31.18317Z","shell.execute_reply":"2022-06-10T13:26:31.187271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple Benchmark\n\nWe use the sample testing instances to compare to my implementation in R.","metadata":{}},{"cell_type":"markdown","source":"### Test 1","metadata":{}},{"cell_type":"code","source":"y_true = pd.DataFrame({'target': [0, 1, 0, 1, 0, 1]})\ny_pred = pd.DataFrame({'prediction': [0.1, 0.9, 0.2, 0.88, 0.3, 0.75]})\n\nprint(f\"top_four_percent_captured: {top_four_percent_captured(y_true, y_pred):.6f}\\n\")\nprint(f\"normalized_weighted_gini: {normalized_weighted_gini(y_true, y_pred):.6f}\\n\")\nprint(f\"amex_metric: {amex_metric(y_true, y_pred):.6f}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-06-10T14:12:55.848141Z","iopub.execute_input":"2022-06-10T14:12:55.849219Z","iopub.status.idle":"2022-06-10T14:12:55.895662Z","shell.execute_reply.started":"2022-06-10T14:12:55.849160Z","shell.execute_reply":"2022-06-10T14:12:55.894719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test 2","metadata":{}},{"cell_type":"code","source":"y_true = pd.DataFrame({'target': [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0]})\ny_pred = pd.DataFrame({'prediction': [0.9, 0.3, 0.8, 0.75, 0.65, 0.6, 0.78, 0.7, 0.05, 0.41, 0.42, 0.05, 0.5, 0.11, 0.12]})\n\nprint(f\"top_four_percent_captured: {top_four_percent_captured(y_true, y_pred):.6f}\\n\")\nprint(f\"normalized_weighted_gini: {normalized_weighted_gini(y_true, y_pred):.6f}\\n\")\nprint(f\"amex_metric: {amex_metric(y_true, y_pred):.6f}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-06-10T14:13:07.534946Z","iopub.execute_input":"2022-06-10T14:13:07.535431Z","iopub.status.idle":"2022-06-10T14:13:07.574955Z","shell.execute_reply.started":"2022-06-10T14:13:07.535399Z","shell.execute_reply":"2022-06-10T14:13:07.574265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test 3","metadata":{}},{"cell_type":"code","source":"y_true = pd.DataFrame({'target': [1, 1, 0, 1, 0, 1, 0, 0, 0, 0]})\ny_pred = pd.DataFrame({'prediction': [0.11, 0.62, 0.61, 0.62, 0.86, 0.64, 0.01, 0.23, 0.67, 0.51]})\n\nprint(f\"top_four_percent_captured: {top_four_percent_captured(y_true, y_pred):.6f}\\n\")\nprint(f\"normalized_weighted_gini: {normalized_weighted_gini(y_true, y_pred):.6f}\\n\")\nprint(f\"amex_metric: {amex_metric(y_true, y_pred):.6f}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-06-10T14:14:15.583462Z","iopub.execute_input":"2022-06-10T14:14:15.583773Z","iopub.status.idle":"2022-06-10T14:14:15.622433Z","shell.execute_reply.started":"2022-06-10T14:14:15.583733Z","shell.execute_reply":"2022-06-10T14:14:15.621781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AMEX Test","metadata":{"execution":{"iopub.status.busy":"2022-06-10T13:48:00.964125Z","iopub.execute_input":"2022-06-10T13:48:00.964437Z","iopub.status.idle":"2022-06-10T13:48:00.968358Z","shell.execute_reply.started":"2022-06-10T13:48:00.964405Z","shell.execute_reply":"2022-06-10T13:48:00.967513Z"}}},{"cell_type":"code","source":"train_data = pd.read_csv(\n    input_path / 'train_data.csv',\n    index_col='customer_ID',\n    usecols=['customer_ID', 'P_2'])\n\ntrain_labels = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ave_p2 = (train_data\n          .groupby('customer_ID')\n          .mean()\n          .rename(columns={'P_2': 'prediction'}))\n\n# Scale the mean P_2 by the max value and take the compliment\nave_p2['prediction'] = 1.0 - (ave_p2['prediction'] / ave_p2['prediction'].max())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric(train_labels, ave_p2)) # 0.572773","metadata":{},"execution_count":null,"outputs":[]}]}