{
  "id": 327162,
  "title": "Amex metric using pd.Series",
  "url": "/competitions/amex-default-prediction/discussion/327162",
  "author_name": "Bruno Gorresen Mello",
  "post_date": "2022-05-25T22:24:53.350000",
  "votes": 9,
  "comment_count": 3,
  "views": 0,
  "content": "<p>I personally prefer when I can calculate the metric by passing the targets and predictions as series because I don't need to name the column the exact same way every time, so I created an alternative version of the function shown <a href=\"https://www.kaggle.com/code/inversion/amex-competition-metric-python\" target=\"_blank\">here</a>:</p>\n<pre><code>def amex_metric(y_true: pd.Series, y_pred: pd.Series) -&gt; float:\n\n    def top_four_percent_captured(df) -&gt; float:\n\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] &lt;= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(df) -&gt; float:\n\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true, df) -&gt; float:\n        y_true_pred = y_true.rename('prediction')\n        true_df = pd.concat([y_true, y_true_pred], axis='columns').sort_values('prediction', ascending=False)\n        return weighted_gini(df) / weighted_gini(true_df)\n\n    df = pd.DataFrame({'target': y_true, 'prediction': y_pred}).sort_values('prediction', ascending=False)\n    g = normalized_weighted_gini(y_true, df.copy())\n    d = top_four_percent_captured(df.copy())\n\n    return 0.5 * (g + d)\n</code></pre>",
  "messages": [
    {
      "id": 1801570,
      "postDate": "2022-05-25T22:24:53.350Z",
      "content": "<p>I personally prefer when I can calculate the metric by passing the targets and predictions as series because I don't need to name the column the exact same way every time, so I created an alternative version of the function shown <a href=\"https://www.kaggle.com/code/inversion/amex-competition-metric-python\" target=\"_blank\">here</a>:</p>\n<pre><code>def amex_metric(y_true: pd.Series, y_pred: pd.Series) -&gt; float:\n\n    def top_four_percent_captured(df) -&gt; float:\n\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] &lt;= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(df) -&gt; float:\n\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true, df) -&gt; float:\n        y_true_pred = y_true.rename('prediction')\n        true_df = pd.concat([y_true, y_true_pred], axis='columns').sort_values('prediction', ascending=False)\n        return weighted_gini(df) / weighted_gini(true_df)\n\n    df = pd.DataFrame({'target': y_true, 'prediction': y_pred}).sort_values('prediction', ascending=False)\n    g = normalized_weighted_gini(y_true, df.copy())\n    d = top_four_percent_captured(df.copy())\n\n    return 0.5 * (g + d)\n</code></pre>",
      "rawMarkdown": "I personally prefer when I can calculate the metric by passing the targets and predictions as series because I don't need to name the column the exact same way every time, so I created an alternative version of the function shown [here](https://www.kaggle.com/code/inversion/amex-competition-metric-python):\n\n    def amex_metric(y_true: pd.Series, y_pred: pd.Series) -> float:\n\n        def top_four_percent_captured(df) -> float:\n\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            four_pct_cutoff = int(0.04 * df['weight'].sum())\n            df['weight_cumsum'] = df['weight'].cumsum()\n            df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n            return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n        def weighted_gini(df) -> float:\n\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n            total_pos = (df['target'] * df['weight']).sum()\n            df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n            df['lorentz'] = df['cum_pos_found'] / total_pos\n            df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n            return df['gini'].sum()\n\n        def normalized_weighted_gini(y_true, df) -> float:\n            y_true_pred = y_true.rename('prediction')\n            true_df = pd.concat([y_true, y_true_pred], axis='columns').sort_values('prediction', ascending=False)\n            return weighted_gini(df) / weighted_gini(true_df)\n\n        df = pd.DataFrame({'target': y_true, 'prediction': y_pred}).sort_values('prediction', ascending=False)\n        g = normalized_weighted_gini(y_true, df.copy())\n        d = top_four_percent_captured(df.copy())\n\n        return 0.5 * (g + d)",
      "votes": 8
    },
    {
      "id": 1882043,
      "postDate": "2022-08-03T03:34:24.980Z",
      "content": "<p>ah! convenient!</p>",
      "rawMarkdown": "ah! convenient!"
    },
    {
      "id": 1807047,
      "postDate": "2022-05-31T16:55:41.070Z",
      "rawMarkdown": "",
      "isDeleted": true
    },
    {
      "id": 1802098,
      "postDate": "2022-05-26T12:57:04.733Z",
      "content": "<p>It is so useful. Thanks!</p>",
      "rawMarkdown": "It is so useful. Thanks!",
      "votes": 1
    }
  ],
  "comments": [
    {
      "id": 1882043,
      "author_name": "Abhinav Boyed",
      "author_url": "",
      "post_date": "2022-08-03T03:34:24.980000",
      "content": "<p>ah! convenient!</p>",
      "votes": 0,
      "replies": []
    },
    {
      "id": 1807047,
      "author_name": "",
      "author_url": "",
      "post_date": "2022-05-31T16:55:41.070000",
      "content": "",
      "votes": 0,
      "replies": []
    },
    {
      "id": 1802098,
      "author_name": "Yue Sun",
      "author_url": "",
      "post_date": "2022-05-26T12:57:04.733000",
      "content": "<p>It is so useful. Thanks!</p>",
      "votes": 1,
      "replies": []
    }
  ],
  "raw_markdown_by_id": {
    "1801570": "I personally prefer when I can calculate the metric by passing the targets and predictions as series because I don't need to name the column the exact same way every time, so I created an alternative version of the function shown [here](https://www.kaggle.com/code/inversion/amex-competition-metric-python):\n\n    def amex_metric(y_true: pd.Series, y_pred: pd.Series) -> float:\n\n        def top_four_percent_captured(df) -> float:\n\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            four_pct_cutoff = int(0.04 * df['weight'].sum())\n            df['weight_cumsum'] = df['weight'].cumsum()\n            df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n            return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n        def weighted_gini(df) -> float:\n\n            df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n            df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n            total_pos = (df['target'] * df['weight']).sum()\n            df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n            df['lorentz'] = df['cum_pos_found'] / total_pos\n            df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n            return df['gini'].sum()\n\n        def normalized_weighted_gini(y_true, df) -> float:\n            y_true_pred = y_true.rename('prediction')\n            true_df = pd.concat([y_true, y_true_pred], axis='columns').sort_values('prediction', ascending=False)\n            return weighted_gini(df) / weighted_gini(true_df)\n\n        df = pd.DataFrame({'target': y_true, 'prediction': y_pred}).sort_values('prediction', ascending=False)\n        g = normalized_weighted_gini(y_true, df.copy())\n        d = top_four_percent_captured(df.copy())\n\n        return 0.5 * (g + d)",
    "1882043": "ah! convenient!",
    "1807047": "",
    "1802098": "It is so useful. Thanks!"
  }
}