{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This R environment comes with many helpful analytics packages installed\n# It is defined by the kaggle/rstats Docker image: https://github.com/kaggle/docker-rstats\n# For example, here's a helpful package to load\n\nlibrary(tidyverse) # metapackage of all tidyverse packages\nlibrary(data.table)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nlist.files(path = \"../input\") \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-05-27T05:20:51.681019Z","iopub.execute_input":"2022-05-27T05:20:51.683625Z","iopub.status.idle":"2022-05-27T05:20:53.293199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### This is simple replication of following notebook in R.\n\nhttps://www.kaggle.com/code/inversion/amex-competition-metric-python","metadata":{}},{"cell_type":"code","source":"amex_metric <- function(target, prediction) {\n  \n  top_four_percent_captured <- function(target, prediction) {\n    \n    dat <- data.frame(target, prediction)\n    \n    dat %>% \n      arrange(-prediction) %>% \n      mutate(weight = case_when(target == 0 ~ 20,\n                                target == 1 ~ 1)) -> dat\n    \n    four_pct_cutoff <- as.integer(0.04 * sum(dat['weight']))\n    dat['cumsum_weight'] <- cumsum(dat$weight)\n    \n    df_cutoff <- dat[dat['cumsum_weight'] <= four_pct_cutoff,]\n    \n    return(sum(df_cutoff['target'] == 1)/sum(dat['target'] == 1))\n  }\n \n  weighted_gini <- function(target, prediction) {\n    \n    dat <- data.frame(target, prediction)\n    dat %>% \n      arrange(-prediction) %>% \n      mutate(weight = case_when(target == 0 ~ 20,\n                                target == 1 ~ 1)) -> dat\n    \n    dat['random'] <- cumsum(dat['weight']/sum(dat['weight']))\n    \n    total_pos <- sum(dat['target'] * dat['weight'])\n    dat['cum_pos_found'] <- cumsum(dat['target'] * dat['weight'])\n    dat['lorentz'] <- dat['cum_pos_found'] / total_pos\n    dat['gini'] <- (dat['lorentz'] - dat['random']) * dat['weight']\n    \n    return(sum(dat['gini']))\n    \n  }\n  \n  normalized_weighted_gini <- function(target, prediction) {\n    return(weighted_gini(target, prediction) / weighted_gini(target, target))\n  }\n  \n  g <- normalized_weighted_gini(target, prediction)\n  d <- top_four_percent_captured(target, prediction)\n  \n  return(0.5 * (g + d))\n  \n}\n","metadata":{"execution":{"iopub.status.busy":"2022-05-27T05:09:21.710458Z","iopub.execute_input":"2022-05-27T05:09:21.712074Z","iopub.status.idle":"2022-05-27T05:09:21.725568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain <- fread(\"../input/amex-default-prediction/train_data.csv\", select = c(\"customer_ID\", 'P_2'))\nnew_tr <- setDT(xtrain)[, lapply(.SD, mean, na.rm=TRUE), keyby = customer_ID] %>% data.frame()\nrm(xtrain)\n\nnew_tr$P_2 <- 1.0 - (new_tr$P_2) / max(new_tr$P_2, na.rm = T)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T05:09:21.935681Z","iopub.execute_input":"2022-05-27T05:09:21.937239Z","iopub.status.idle":"2022-05-27T05:10:48.520707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ytrain <- fread(\"../input/amex-default-prediction/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-05-27T05:10:48.523503Z","iopub.execute_input":"2022-05-27T05:10:48.52513Z","iopub.status.idle":"2022-05-27T05:10:49.241438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(ytrain$target, new_tr$P_2)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T05:11:01.547862Z","iopub.execute_input":"2022-05-27T05:11:01.549732Z","iopub.status.idle":"2022-05-27T05:11:02.009538Z"},"trusted":true},"execution_count":null,"outputs":[]}]}