{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30618,"isInternetEnabled":false,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"###############################\n# Test/Dummy Submission\n###############################\n\n# random number submission or similar\n# create predictions on the test and submit\n# shouts out\n    # https://www.kaggle.com/code/gyogyocat/a-beginner-s-quick-eda-in-r\n    # https://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook\n\n# load packages\nsuppressPackageStartupMessages(library(tidyverse))\nsuppressPackageStartupMessages(library(summarytools))\nsuppressPackageStartupMessages(library(inspectdf))\n                               \n# if needed:\n# suppressPackageStartupMessages(library(data.table)) \n# suppressPackageStartupMessages(library(dtplyr)) \n\n# avaiable files...\n# later, joining/harmonizing many datasets will be a challenge\ndir(\"..\")\ncat('\\n --------- \\n available files: TOP LEVEL \\n')\nlist.files(path = \"../input/home-credit-credit-risk-model-stability\") %>% print()\ncat('\\n --------- \\n CSVs \\n')\nlist.files(path = \"../input/home-credit-credit-risk-model-stability/csv_files\") %>% print()\ncat('\\n --------- \\n TRAINING FILES \\n')\nlist.files(path = \"../input/home-credit-credit-risk-model-stability/csv_files/train\") %>% print()\ncat('\\n --------- \\n TEST FILES \\n')\nlist.files(path = \"../input/home-credit-credit-risk-model-stability/csv_files/test\") %>% print()","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2024-02-20T03:13:04.653874Z","iopub.execute_input":"2024-02-20T03:13:04.659674Z","iopub.status.idle":"2024-02-20T03:13:04.787095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the base table from a CSV file\ntrain_basetable <- read.csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\") %>% arrange(case_id)\nglimpse(train_basetable)\ncat('---------\\n')\n#\n# T/F: are these individual cases?\nprint('are these individual cases?')\nprint(nrow(train_basetable) == nrow( train_basetable %>% distinct(case_id)) )\ncat('---------\\n')\n#\nprint('distribution of target')\nggplot(data=train_basetable)+geom_bar(aes(x=factor(target))) + theme(text=element_text(size=20))\ntrain_basetable %>% count(target) %>% mutate(perc=n/sum(n))","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:15:38.619513Z","iopub.execute_input":"2024-02-20T03:15:38.621510Z","iopub.status.idle":"2024-02-20T03:15:44.607776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load 'static' files\ntrain_static_0_0 <- read.csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_0.csv\") %>% arrange(case_id)\ntrain_static_0_1 <- read.csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_1.csv\") %>% arrange(case_id)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T02:09:41.140464Z","iopub.execute_input":"2024-02-20T02:09:41.142420Z","iopub.status.idle":"2024-02-20T02:12:06.446584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view static 0_0\nglimpse(train_static_0_0)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T02:12:06.585953Z","iopub.execute_input":"2024-02-20T02:12:06.588097Z","iopub.status.idle":"2024-02-20T02:12:06.762645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view static 0_1\nglimpse(train_static_0_1)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T02:12:06.766364Z","iopub.execute_input":"2024-02-20T02:12:06.768235Z","iopub.status.idle":"2024-02-20T02:12:06.983257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# T/F: are the static datasets 1-to-1 with training basetable?\nnrow(train_basetable) == nrow(train_static_0_0)+nrow(train_static_0_1)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T02:12:06.990412Z","iopub.execute_input":"2024-02-20T02:12:06.993447Z","iopub.status.idle":"2024-02-20T02:12:07.032755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# T/F: 0_0 and 0_1 means we can stack without problems?\ncolnames(train_static_0_0) == colnames(train_static_0_1)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T02:48:47.954804Z","iopub.execute_input":"2024-02-20T02:48:47.957394Z","iopub.status.idle":"2024-02-20T02:48:47.982537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# next static CB tables\ntrain_static_cb <- read.csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_cb_0.csv\") %>% arrange(case_id)\nglimpse(train_static_cb)\ncat('\\n ------------ NOTE: this is close, but does not have the same number of cases as training base file...')\nprint(nrow(train_basetable) == nrow(train_static_cb))","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:20:26.367072Z","iopub.execute_input":"2024-02-20T03:20:26.369098Z","iopub.status.idle":"2024-02-20T03:20:50.112849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n\n**** ALRIGHT, WE CAN STACK/JOIN THOSE AND PLAY AROUND LATER. FOR NOW, IGNORING FEATURES TO CREATE A TEST SUBMISSION.****\n\n\n\n","metadata":{}},{"cell_type":"code","source":"# Read the base table from a CSV file\ntest_basetable <- read.csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv\")\nglimpse(test_basetable)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:22:41.263241Z","iopub.execute_input":"2024-02-20T03:22:41.265363Z","iopub.status.idle":"2024-02-20T03:22:41.310748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n**** OK SO THEY ARE BASICALLY SHOWING US NOTHING ****\n\n","metadata":{}},{"cell_type":"code","source":"# still, useful to know structure of test_base\nhead(test_basetable, 20)\ncat('\\n ---- \\n')\nprint('T/F: are these individual cases?')\nprint(nrow(test_basetable) == nrow( test_basetable %>% distinct(case_id)) )","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:29:58.328303Z","iopub.execute_input":"2024-02-20T03:29:58.330513Z","iopub.status.idle":"2024-02-20T03:29:58.386088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**** BELOW: A SIMPLE PROCEDURE FOR A DUMMY SUBMISSION ****\n","metadata":{}},{"cell_type":"code","source":"test_length <- nrow(test_basetable)\ntest_length","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:32:11.798980Z","iopub.execute_input":"2024-02-20T03:32:11.802137Z","iopub.status.idle":"2024-02-20T03:32:11.840919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#nrows in test data\ntest_length <- nrow(test_basetable)\n\n#simulate random vector of 0s and 1s\nbenchmark <- rbinom(test_length, 1, 0.05)\n#convert it to 0.95 or 0.05\nbenchmark <- benchmark*0.95 + (1-benchmark)*0.05\n\n#checks\nlength(benchmark)\nhead(benchmark, 15)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:34:52.033936Z","iopub.execute_input":"2024-02-20T03:34:52.036281Z","iopub.status.idle":"2024-02-20T03:34:52.080610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#bind case_id and predictions\ntest_df <- tibble(\"case_id\"=test_basetable$case_id, \"score\"=benchmark)\nnrow(test_df)\nhead(test_df, 15)","metadata":{"execution":{"iopub.status.busy":"2024-02-20T03:37:14.739546Z","iopub.execute_input":"2024-02-20T03:37:14.741550Z","iopub.status.idle":"2024-02-20T03:37:14.800569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#write to submission.csv \nwrite_csv(x=test_df, file=\"submission.csv\")","metadata":{},"execution_count":null,"outputs":[]}]}