{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Test/dummy submission notebook\n# Upload only the data necessary to create a random vector of predictions and submit\n\n#Data size is the main problem...\n#things I looked at:\n    # https://www.kaggle.com/competitions/amex-default-prediction/discussion/328054\n    #https://www.kaggle.com/code/nikhilsharma24/custom-metric-r/comments\n\nsuppressPackageStartupMessages(library(data.table)) \nsuppressPackageStartupMessages(library(tidyverse))\nsuppressPackageStartupMessages(library(dtplyr))\n\ndir(\"..\")\nprint('available files...')\nlist.files(path = \"../input/amex-default-prediction\") %>% print()","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-06-07T13:44:20.509480Z","iopub.execute_input":"2022-06-07T13:44:20.598526Z","iopub.status.idle":"2022-06-07T13:44:20.668526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#look at sample submission\nsample_submission <- fread(\"../input/amex-default-prediction/sample_submission.csv\") %>% as_tibble()\n\nprint(paste('Number of rows',nrow(sample_submission)))\nprint(paste('Number of distinct IDs',nrow(distinct(sample_submission))))\nprint('data format...')\nhead(sample_submission)\nprint('distribution of outcomes...')\nggplot(data=sample_submission)+geom_histogram(aes(x=prediction))","metadata":{"execution":{"iopub.status.busy":"2022-06-07T13:48:02.706914Z","iopub.execute_input":"2022-06-07T13:48:02.709868Z","iopub.status.idle":"2022-06-07T13:48:04.613984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find number of rows, structure of test data\nt1 <- Sys.time()\ntest_ids <- fread(\"../input/amex-default-prediction/test_data.csv\", select=c(1L,2L)) %>% as_tibble()\nt2 <- Sys.time()\n#first_run: 241.2295 secs\n\nprint('Time to load first 2 cols')\ndifftime(t2,t1, units=\"secs\")\nprint('number of rows...')\ntest_ids %>% nrow()\nprint('number of customers...')\ntest_ids %>% distinct(customer_ID) %>% nrow()\nprint('number of unique dates...')\ntest_ids %>% distinct(S_2) %>% nrow()\nprint('data format...')\ntest_ids %>% print(n=5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#something simple to base the predictions off of\n#find proportion of 0s and 1s in target\nt1 <- Sys.time()\ntrain_labels <- fread(\"../input/amex-default-prediction/train_labels.csv\") %>% as_tibble()\nt2 <- Sys.time()\n\nprint('Time to load training labels..')\ndifftime(t2,t1, units=\"secs\")\nprint('Number of rows')\ntrain_labels %>% nrow()\nprint('Number of IDs')\ntrain_labels %>% distinct(customer_ID) %>% nrow()\nprint('Proportion of defaults')\ntrain_labels %>% count(target) %>% mutate(perc = n/sum(n)) %>% print()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get test ids for prediction\nt1 <- Sys.time()\ntest_pred_labels <- test_ids %>% distinct(customer_ID)\nt2 <- Sys.time()\n\nprint('Time to get the distinct customer ids...')\ndifftime(t2,t1, units=\"secs\")\n\n#number of vals to predict \ntest_length <- nrow(test_pred_labels)\n\nhead(test_pred_labels)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#simulate random vector of 0s and 1s\nbenchmark <- rbinom(test_length, 1, 0.259)\n#convert it to 0.25 or 0.75\nbenchmark <- benchmark*0.75 + (1-benchmark)*0.25\n\n#checks\nlength(benchmark)\nhead(benchmark, 15)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#bind test_IDs and predictions\n#library(tidyverse)\ntest_df <- tibble(test_pred_labels,prediction=benchmark)\nnrow(test_df)\nhead(test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#write to submission.csv \nfwrite(x=test_df, file=\"submission.csv\")","metadata":{},"execution_count":null,"outputs":[]}]}