{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is intended for those who want a gentle introduction to the Evaluation Metric for this competition. \n\nI created this notebook to decomose the Evaluation Metric, M, and its components, D and G.","metadata":{}},{"cell_type":"code","source":"import pandas as pd","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's create a simple benchmark to calculate the Evaluation Metric. We are going to use *P_2* column in train_data.csv to calculate *y_pred* and train_labels.csv to create *y_true*.","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/amex-default-prediction/train_data.csv', index_col='customer_ID', usecols=['customer_ID', 'P_2'])\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv', index_col='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:35:45.300444Z","iopub.execute_input":"2022-05-27T10:35:45.300920Z","iopub.status.idle":"2022-05-27T10:40:08.831442Z","shell.execute_reply.started":"2022-05-27T10:35:45.300885Z","shell.execute_reply":"2022-05-27T10:40:08.827876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:48:54.610301Z","iopub.execute_input":"2022-05-27T10:48:54.611754Z","iopub.status.idle":"2022-05-27T10:48:54.646018Z","shell.execute_reply.started":"2022-05-27T10:48:54.611683Z","shell.execute_reply":"2022-05-27T10:48:54.645009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:48:56.148382Z","iopub.execute_input":"2022-05-27T10:48:56.149729Z","iopub.status.idle":"2022-05-27T10:48:56.159287Z","shell.execute_reply.started":"2022-05-27T10:48:56.149678Z","shell.execute_reply":"2022-05-27T10:48:56.158599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there are a couple of rows for each *customer_id*, we are going to group them and then calculate the mean of *P_2* column as our prediction.","metadata":{}},{"cell_type":"code","source":"ave_p2 = (train_data.groupby('customer_ID').mean().rename(columns={'P_2': 'prediction'}))\n\n# Scale the mean P_2 by the max value and take the compliment\nave_p2['prediction'] = 1.0 - (ave_p2['prediction'] / ave_p2['prediction'].max())\n\nave_p2.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:50:48.377029Z","iopub.execute_input":"2022-05-27T10:50:48.377506Z","iopub.status.idle":"2022-05-27T10:50:50.126207Z","shell.execute_reply.started":"2022-05-27T10:50:48.377471Z","shell.execute_reply":"2022-05-27T10:50:50.125536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = train_labels.copy()\ny_pred = ave_p2.copy()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:50:59.103421Z","iopub.execute_input":"2022-05-27T10:50:59.103830Z","iopub.status.idle":"2022-05-27T10:50:59.110857Z","shell.execute_reply.started":"2022-05-27T10:50:59.103789Z","shell.execute_reply":"2022-05-27T10:50:59.110165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The evaluation metric, **M**, for this competition is the mean of two measures of rank ordering: Normalized Gini Coefficient, **G**, and default rate captured at 4%, **D**.\n\n*M = 0.5 x (G + D)*\n\nIn the following, we will calculate each component in detail.","metadata":{}},{"cell_type":"markdown","source":"##### Calculating Parameter D (default rate captured at 4%):\nThe default rate captured at 4% is the percentage of the positive labels (defaults) captured within the highest-ranked 4% of the predictions, and represents a Sensitivity/Recall statistic.\n\nLet's break down the steps to caluclate D:","metadata":{}},{"cell_type":"code","source":"# Create a df dataframe by concatinating y_true, y_pred and sorting the rows based on y_pred in a descending order.\ndf = (pd.concat([y_true, y_pred], axis='columns').sort_values('prediction', ascending=False))\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:53:32.393221Z","iopub.execute_input":"2022-05-27T10:53:32.393687Z","iopub.status.idle":"2022-05-27T10:53:32.769620Z","shell.execute_reply.started":"2022-05-27T10:53:32.393653Z","shell.execute_reply":"2022-05-27T10:53:32.768603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a 'weight' column in df with values of 1 for y_true = 0, and 20 for y_true = 0.\ndf['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:53:52.951784Z","iopub.execute_input":"2022-05-27T10:53:52.952221Z","iopub.status.idle":"2022-05-27T10:53:53.258253Z","shell.execute_reply.started":"2022-05-27T10:53:52.952188Z","shell.execute_reply":"2022-05-27T10:53:53.257426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate four_pct_cutoff variable as 4 percent of the sum of all values in 'weight' column.\nfour_pct_cutoff = int(0.04 * df['weight'].sum())\nfour_pct_cutoff","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:54:34.767899Z","iopub.execute_input":"2022-05-27T10:54:34.768277Z","iopub.status.idle":"2022-05-27T10:54:34.775703Z","shell.execute_reply.started":"2022-05-27T10:54:34.768248Z","shell.execute_reply":"2022-05-27T10:54:34.775015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'weight_cumsum' column in df to calculate cumulative sum of weights in 'weight' column.\ndf['weight_cumsum'] = df['weight'].cumsum()\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:54:56.850146Z","iopub.execute_input":"2022-05-27T10:54:56.850627Z","iopub.status.idle":"2022-05-27T10:54:56.870856Z","shell.execute_reply.started":"2022-05-27T10:54:56.850583Z","shell.execute_reply":"2022-05-27T10:54:56.869618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create df_cutoff dataframe based on filtering the portion of descending sorted y_pred which has lower weight_cumsum of four_pct_cutoff.\ndf_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\ndf_cutoff","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:55:31.773432Z","iopub.execute_input":"2022-05-27T10:55:31.773820Z","iopub.status.idle":"2022-05-27T10:55:31.806846Z","shell.execute_reply.started":"2022-05-27T10:55:31.773790Z","shell.execute_reply":"2022-05-27T10:55:31.805758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the ratio of the number of y_true = 1 in df_cutoff to number of y_true = 1 in the input dataframe.\nd = (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\nd","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:03:32.785622Z","iopub.execute_input":"2022-05-27T11:03:32.786053Z","iopub.status.idle":"2022-05-27T11:03:32.795480Z","shell.execute_reply.started":"2022-05-27T11:03:32.786018Z","shell.execute_reply":"2022-05-27T11:03:32.794539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Calculating Weighted Gini:\n\nWeighted Gini is required to calculate normalized weighted Gini (G parameter).\n\nLet's break down the steps to caluclate Weighted Gini:","metadata":{}},{"cell_type":"code","source":"# Create df dataframe by concatinating y_true, y_pred and sorting the rows based on y_pred in a descending order.\ndf = (pd.concat([y_true, y_pred], axis='columns').sort_values('prediction', ascending=False))\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:58:38.602625Z","iopub.execute_input":"2022-05-27T10:58:38.603685Z","iopub.status.idle":"2022-05-27T10:58:38.927510Z","shell.execute_reply.started":"2022-05-27T10:58:38.603632Z","shell.execute_reply":"2022-05-27T10:58:38.926506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'weight' column in df with values of 1 for y_true = 0, and 20 for y_true = 0.\ndf['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:59:05.450064Z","iopub.execute_input":"2022-05-27T10:59:05.451122Z","iopub.status.idle":"2022-05-27T10:59:05.761089Z","shell.execute_reply.started":"2022-05-27T10:59:05.451053Z","shell.execute_reply":"2022-05-27T10:59:05.760089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'random' column in df that has cumulative sum of 'weight' column divided by sum of 'weight' column.\ndf['random'] = (df['weight'] / df['weight'].sum()).cumsum()\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T10:59:25.526990Z","iopub.execute_input":"2022-05-27T10:59:25.527584Z","iopub.status.idle":"2022-05-27T10:59:25.559061Z","shell.execute_reply.started":"2022-05-27T10:59:25.527534Z","shell.execute_reply":"2022-05-27T10:59:25.558012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define total_pos variable by the sum of all y_true values to their corresponding 'weight' values.\ntotal_pos = (df['target'] * df['weight']).sum()\ntotal_pos","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:00:09.019565Z","iopub.execute_input":"2022-05-27T11:00:09.020053Z","iopub.status.idle":"2022-05-27T11:00:09.030888Z","shell.execute_reply.started":"2022-05-27T11:00:09.020019Z","shell.execute_reply":"2022-05-27T11:00:09.030003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'cum_pos_found' column in df by calculating cumulative sum of the multiplication of y_true and their corresponding 'weight' values.\ndf['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:00:39.434084Z","iopub.execute_input":"2022-05-27T11:00:39.434567Z","iopub.status.idle":"2022-05-27T11:00:39.457711Z","shell.execute_reply.started":"2022-05-27T11:00:39.434531Z","shell.execute_reply":"2022-05-27T11:00:39.456534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'lorentz' column by dividing 'cum_pos_found' values by total_pos value.\ndf['lorentz'] = df['cum_pos_found'] / total_pos\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:01:08.165079Z","iopub.execute_input":"2022-05-27T11:01:08.165523Z","iopub.status.idle":"2022-05-27T11:01:08.189778Z","shell.execute_reply.started":"2022-05-27T11:01:08.165491Z","shell.execute_reply":"2022-05-27T11:01:08.188658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 'gini' column by ('lorentz' - 'random') * 'weight' values\ndf['gini'] = (df['lorentz'] - df['random']) * df['weight']\ndf","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:01:29.889891Z","iopub.execute_input":"2022-05-27T11:01:29.890431Z","iopub.status.idle":"2022-05-27T11:01:29.922367Z","shell.execute_reply.started":"2022-05-27T11:01:29.890391Z","shell.execute_reply":"2022-05-27T11:01:29.921076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate sum of 'gini' column values.\ng_not_normalized = df['gini'].sum()\ng_not_normalized","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:05:24.575900Z","iopub.execute_input":"2022-05-27T11:05:24.576481Z","iopub.status.idle":"2022-05-27T11:05:24.586431Z","shell.execute_reply.started":"2022-05-27T11:05:24.576443Z","shell.execute_reply":"2022-05-27T11:05:24.585596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's put everything togheter in two functions to calculate **d** and **g_not_normalized**:","metadata":{}},{"cell_type":"code","source":"def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \ndef weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:09:06.994034Z","iopub.execute_input":"2022-05-27T11:09:06.994630Z","iopub.status.idle":"2022-05-27T11:09:07.009617Z","shell.execute_reply.started":"2022-05-27T11:09:06.994584Z","shell.execute_reply":"2022-05-27T11:09:07.008318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('D = {}'.format(top_four_percent_captured(y_true, y_pred)))\nprint('g_not_normalized = {}'.format(weighted_gini(y_true, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:11:00.684405Z","iopub.execute_input":"2022-05-27T11:11:00.684800Z","iopub.status.idle":"2022-05-27T11:11:02.088763Z","shell.execute_reply.started":"2022-05-27T11:11:00.684769Z","shell.execute_reply":"2022-05-27T11:11:02.087247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Calculating G (Normalized Weighted Gini):\n\nLet's break down the steps to caluclate Normalized Weighted Gini:\n- Create y_true_pred dataframe by modifying 'target' column name to 'prediction' column name in y_true dataframe.\n- Return weighted_gini(y_true, y_pred) divided by weighted_gini(y_true, y_true_pred)","metadata":{}},{"cell_type":"code","source":"y_true_pred = y_true.rename(columns={'target': 'prediction'})\ny_true_pred","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:05:52.606826Z","iopub.execute_input":"2022-05-27T11:05:52.607470Z","iopub.status.idle":"2022-05-27T11:05:52.628454Z","shell.execute_reply.started":"2022-05-27T11:05:52.607418Z","shell.execute_reply":"2022-05-27T11:05:52.627268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:11:17.466122Z","iopub.execute_input":"2022-05-27T11:11:17.466573Z","iopub.status.idle":"2022-05-27T11:11:18.585682Z","shell.execute_reply.started":"2022-05-27T11:11:17.466540Z","shell.execute_reply":"2022-05-27T11:11:18.584408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's put everything togheter in one functions to calculate **g_normalized**:","metadata":{}},{"cell_type":"code","source":"def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    y_true_pred = y_true.rename(columns={'target': 'prediction'})\n    return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:12:09.019818Z","iopub.execute_input":"2022-05-27T11:12:09.020227Z","iopub.status.idle":"2022-05-27T11:12:09.028006Z","shell.execute_reply.started":"2022-05-27T11:12:09.020195Z","shell.execute_reply":"2022-05-27T11:12:09.026070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have calculated **D** and **G**, we can measure the evaluation metric, **M**.","metadata":{}},{"cell_type":"code","source":"g = normalized_weighted_gini(y_true, y_pred)\nd = top_four_percent_captured(y_true, y_pred)\n\n0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:12:35.449673Z","iopub.execute_input":"2022-05-27T11:12:35.450097Z","iopub.status.idle":"2022-05-27T11:12:37.052781Z","shell.execute_reply.started":"2022-05-27T11:12:35.450065Z","shell.execute_reply":"2022-05-27T11:12:37.051838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's put everything in one place:","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:12:54.796651Z","iopub.execute_input":"2022-05-27T11:12:54.797070Z","iopub.status.idle":"2022-05-27T11:12:54.813966Z","shell.execute_reply.started":"2022-05-27T11:12:54.797038Z","shell.execute_reply":"2022-05-27T11:12:54.812644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:12:57.728549Z","iopub.execute_input":"2022-05-27T11:12:57.729068Z","iopub.status.idle":"2022-05-27T11:12:59.419465Z","shell.execute_reply.started":"2022-05-27T11:12:57.729022Z","shell.execute_reply":"2022-05-27T11:12:59.418259Z"},"trusted":true},"execution_count":null,"outputs":[]}]}