{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This R environment comes with many helpful analytics packages installed\n# It is defined by the kaggle/rstats Docker image: https://github.com/kaggle/docker-rstats\n# For example, here's a helpful package to load\n\nlibrary(tidyverse) # metapackage of all tidyverse packages\nlibrary(data.table)\nlibrary(ggplot2)\n#library(arrow)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nlist.files(path = \"../input\")\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\noptions(repr.plot.width = 12, repr.plot.height = 10)\noptions(warn=-1)","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-06-10T06:24:40.350758Z","iopub.execute_input":"2022-06-10T06:24:40.352945Z","iopub.status.idle":"2022-06-10T06:24:42.001182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain <- fread('../input/amex-default-prediction/train_data.csv', na.strings = c(\"\",NA))\nytrain <- read.csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:24:42.003604Z","iopub.execute_input":"2022-06-10T06:24:42.036347Z","iopub.status.idle":"2022-06-10T06:27:11.551652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only risk variables\n\n#xtrain %>% \n#  select('customer_ID', 'S_2', contains('R_')) -> xtrain\n\n#train <- left_join(xtrain, ytrain)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:11.555109Z","iopub.execute_input":"2022-06-10T06:27:11.557088Z","iopub.status.idle":"2022-06-10T06:27:11.579317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we have 5 types of variables\n\n#D_* = Delinquency variables\n#S_* = Spend variables\n#P_* = Payment variables\n#B_* = Balance variables\n#R_* = Risk variables\n\n\nfact_var <- c('B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68')\nnum_var <- setdiff(names(xtrain), c(\"customer_ID\" ,\"S_2\", fact_var))\n\n# function to select one type of variable for explorationn\n\nselect_type_var <- function(Df, type) {\n  \n  xtrain %>%\n    select('customer_ID', num_var) %>% \n    select('customer_ID', contains(type)) -> df\n  \n  return(df)\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:11.582239Z","iopub.execute_input":"2022-06-10T06:27:11.584060Z","iopub.status.idle":"2022-06-10T06:27:11.603055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"select_type_var(xtrain, 'R') -> xtrain","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:11.605600Z","iopub.execute_input":"2022-06-10T06:27:11.607017Z","iopub.status.idle":"2022-06-10T06:27:11.707782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train <- left_join(xtrain, ytrain)\n\n# convert customer_ID into integer\ntrain[, customer_ID := as.integer(.GRP), by = customer_ID ]\n\n# create new index column for each customer \ntrain[, index := as.integer(row.names(.SD)), by = customer_ID]","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:11.710285Z","iopub.execute_input":"2022-06-10T06:27:11.711786Z","iopub.status.idle":"2022-06-10T06:27:24.090240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fun_plot <- function(Data, Variable, random_chose_ID) {\n  \n  unique_ids <- length(unique(train$customer_ID))\n  random_id <- sample.int(unique_ids, random_chose_ID, replace = F)\n  \n  Data %>% filter(customer_ID %in% random_id) -> plot_dat\n  \n  ggplot(plot_dat, aes(x = index, y = {{Variable}}, group = customer_ID, color = as.factor(target))) +\n    geom_line() +\n    labs(color = \"Target\") +\n    theme(axis.text = element_text(size=25),\n          axis.title = element_text(size=25,face=\"bold\"),\n          legend.title = element_text(size=20),\n          legend.text = element_text(size=20)) -> ts_plot\n  \n  \n  plot_dat %>%\n    select(customer_ID, target) %>%\n    filter(customer_ID %in% random_id) %>%\n    group_by(customer_ID) %>%\n    summarise(n = mean(target)) %>%\n    select(n) %>%\n    table() -> tab\n  \n  return(list(distribution_of_target = tab, \n               ts_plots = ts_plot))\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:24.093022Z","iopub.execute_input":"2022-06-10T06:27:24.094592Z","iopub.status.idle":"2022-06-10T06:27:24.110318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot of randomly chosen 10000 ID's\n\nfun_plot(train, R_1, 10000)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:24.112817Z","iopub.execute_input":"2022-06-10T06:27:24.114260Z","iopub.status.idle":"2022-06-10T06:27:38.125206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fun_plot(train, R_3, 10000)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T06:27:38.127743Z","iopub.execute_input":"2022-06-10T06:27:38.129250Z","iopub.status.idle":"2022-06-10T06:27:48.262691Z"},"trusted":true},"execution_count":null,"outputs":[]}]}