{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Original is https://www.kaggle.com/code/dkraynak/amex-basic-logit-with-feather-data by @dkraynak.\n\nAll I did was just stop converting the predictions to binary values.\nRanking is very important for the metric of this competition, so it is better to submit the predictions as probabilities.","metadata":{}},{"cell_type":"code","source":"# Basic Logit Submission\n# Working with feather data, but want to see how far basic tools can get me\n# Still have to be very careful with memory, can't really handle test data in memory (even as feather)\n# Started with: https://www.kaggle.com/code/stautxie/amex-default-prediction-model-in-r-part-1\n\nsuppressPackageStartupMessages(library(data.table)) \nsuppressPackageStartupMessages(library(tidyverse))\nsuppressPackageStartupMessages(library(dtplyr)) #data.table with tidy syntax\nsuppressPackageStartupMessages(library(arrow))\n\ndir(\"..\")\nprint('available files...')\nlist.files(path = \"../input/amex-default-prediction\") %>% print()\n\npqt_dir <- '../input/amex-data-integer-dtypes-parquet-format'\ncsv_dir <- '../input/amex-default-prediction'\ndt_threads <- getDTthreads()\ncat(paste(\"Number of threads for data.table: \", dt_threads, \"\\n\", sep=\"\"))","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-07-06T00:50:31.858839Z","iopub.execute_input":"2022-07-06T00:50:31.861053Z","iopub.status.idle":"2022-07-06T00:50:31.908988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pull in training labels\nt1 <- Sys.time()\ntrain_Y <- fread(\"../input/amex-default-prediction/train_labels.csv\") %>% as_tibble()\nt2 <- Sys.time()\n\nprint('Time to load training labels..')\ndifftime(t2,t1, units=\"secs\")\nprint('Number of rows')\ntrain_Y %>% nrow()\nprint('Number of IDs')\ntrain_Y %>% distinct(customer_ID) %>% nrow()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:50:31.912222Z","iopub.execute_input":"2022-07-06T00:50:31.914066Z","iopub.status.idle":"2022-07-06T00:50:32.851985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load subset of the efficiently-stored training data, collapse to 1 row customer on import\nt1 <- Sys.time()\n\ntrain_df <- \n    arrow::read_parquet(file.path(pqt_dir, \"train.parquet\"), col_select = 1:10) %>% \n    mutate(S_2 = lubridate::ymd(S_2)) %>%\n    group_by(customer_ID) %>% \n    slice_max(S_2) %>% \n    ungroup()\n\nt2 <- Sys.time()\n\nprint('Time to load training parquet file with arrow and grab latest obs per group')\ndifftime(t2,t1, units=\"secs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:50:32.854499Z","iopub.execute_input":"2022-07-06T00:50:32.85594Z","iopub.status.idle":"2022-07-06T00:53:42.875002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#attach training labels to training data\ntrain_df <- left_join(train_Y,train_df,by=c(\"customer_ID\"))\n\nnrow(train_df)\ncolSums(is.na(train_df))\n\nprint('Proportion of defaults')\ntrain_df %>% count(target) %>% mutate(perc = n/sum(n)) %>% print()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:53:42.877372Z","iopub.execute_input":"2022-07-06T00:53:42.878725Z","iopub.status.idle":"2022-07-06T00:53:43.16243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#estimate logit on the first few variables\nmodel1 = glm(target ~ P_2 + B_1 + B_2 + R_1 + D_39 + S_3, data=train_df, family=\"binomial\")\nsummary(model1)\n\n#predict\npredictions = predict(model1, newdata=train_df, type=\"response\") %>% replace_na(0) #fallback should definitely be a 0\nhead(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:53:43.165018Z","iopub.execute_input":"2022-07-06T00:53:43.166561Z","iopub.status.idle":"2022-07-06T00:53:45.337658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remotes::install_github(\"igjit/amexmetric\", quiet = TRUE)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:54:07.316676Z","iopub.execute_input":"2022-07-06T00:54:07.319564Z","iopub.status.idle":"2022-07-06T00:54:07.486873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amexmetric::amex_metric(pull(train_Y,target), predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T00:54:13.742709Z","iopub.execute_input":"2022-07-06T00:54:13.744381Z","iopub.status.idle":"2022-07-06T00:54:14.280666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tidy up workspace (hopefully will be able to keep the model and the test data in memory at the same time...)\nrm(train_df,train_Y,predictions)\ngc()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:02:06.031736Z","iopub.execute_input":"2022-07-06T01:02:06.034073Z","iopub.status.idle":"2022-07-06T01:02:06.968636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the efficiently-stored test data, collapse to 1 row per customer on import\nt1 <- Sys.time()\n\ntest_X <- \n    arrow::read_parquet(file.path(pqt_dir, \"test.parquet\"), col_select = 1:10) %>% \n    mutate(S_2 = lubridate::ymd(S_2)) %>%\n    group_by(customer_ID) %>% \n    slice_max(S_2) %>% \n    ungroup()\n\nt2 <- Sys.time()\n\nprint('Time to load test parquet file with arrow and grab latest obs per group')\ndifftime(t2,t1, units=\"secs\")\n\nclass(test_X)\nnrow(test_X)\n#test_X","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:02:09.464542Z","iopub.execute_input":"2022-07-06T01:02:09.466169Z","iopub.status.idle":"2022-07-06T01:09:18.508152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make predictions on test data\n\n#predict\npredictions = predict(model1, newdata=test_X, type=\"response\") %>% replace_na(0) #fallback should definitely be a 0\nhead(predictions)\n\n#check distribution\nhist(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:09:31.488565Z","iopub.execute_input":"2022-07-06T01:09:31.490289Z","iopub.status.idle":"2022-07-06T01:09:32.846212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#bind test_IDs and predictions\ntest_df <- test_X %>% mutate(prediction=predictions) %>% select(customer_ID,prediction)\nnrow(test_df)\nhead(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:09:46.023352Z","iopub.execute_input":"2022-07-06T01:09:46.025205Z","iopub.status.idle":"2022-07-06T01:09:46.406099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#write to submission.csv \nfwrite(x=test_df, file=\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T01:10:01.573262Z","iopub.execute_input":"2022-07-06T01:10:01.574923Z","iopub.status.idle":"2022-07-06T01:10:02.003293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}