{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"library(tidyverse)","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2021-08-03T17:00:17.211720Z","iopub.execute_input":"2021-08-03T17:00:17.215219Z","iopub.status.idle":"2021-08-03T17:00:18.589364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list.files(\"../input/kddbr-2021/\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:03:32.373771Z","iopub.execute_input":"2021-08-03T17:03:32.375325Z","iopub.status.idle":"2021-08-03T17:03:32.399649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We provide a file that contains the information about the sizes of each graph in the test set.\nWe use this information to keep only the possible edges in the submission files (that is, removing self-loops and edges that are not supposed to exist.)\n\n*This information can also be inferred from the masks in the training/test feature datasets.*","metadata":{}},{"cell_type":"code","source":"size_data <- read_csv(\"../input/kddbr-2021/TSP2Ksizes.csv\", col_types = cols(.default = col_integer()))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:11:51.599413Z","iopub.execute_input":"2021-08-03T17:11:51.600922Z","iopub.status.idle":"2021-08-03T17:11:51.623223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The file `zeroes_output.csv` is in the same format as `TSP100Ktarget.csv`.  Note that, in these files, there are 40K columns representing a 200 x 200 adjacency matrix. Many of the values in this format have no meaning and will be discarded in the submission (self-loops and extraneous edges in smaller graphs).","metadata":{}},{"cell_type":"code","source":"data <- read_csv(\"../input/kddbr-2021/zeroes_output.csv\", col_types = cols(.default = col_integer()))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:04:00.873376Z","iopub.execute_input":"2021-08-03T17:04:00.875150Z","iopub.status.idle":"2021-08-03T17:04:25.279773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Joining size data","metadata":{}},{"cell_type":"code","source":"data <- data %>% left_join(size_data, by = \"graph\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:04:25.282070Z","iopub.execute_input":"2021-08-03T17:04:25.283456Z","iopub.status.idle":"2021-08-03T17:04:39.842634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Tranforming \"wide\" to \"long\"","metadata":{}},{"cell_type":"code","source":"data <- data %>% pivot_longer(-c(graph, size), names_to = \"edge\", values_to = \"value\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:04:39.844962Z","iopub.execute_input":"2021-08-03T17:04:39.846326Z","iopub.status.idle":"2021-08-03T17:04:51.484593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Extracting edge endpoints (from and to)","metadata":{}},{"cell_type":"code","source":"data <- data %>% \n    mutate(edge = str_sub(edge, 3)) %>%\n    mutate(from = as.numeric(str_replace(edge, \"[^,]+,\", \"\"))) %>% \n    mutate(to = as.numeric(str_replace(edge, \",.*\", \"\"))) %>% \n    select(-edge)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:05:22.405822Z","iopub.execute_input":"2021-08-03T17:05:22.407425Z","iopub.status.idle":"2021-08-03T17:07:28.963437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Removing irrelevant edge predictions","metadata":{}},{"cell_type":"code","source":"data <- data %>% filter(from < size & to < size & from != to)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:07:28.965722Z","iopub.execute_input":"2021-08-03T17:07:28.967053Z","iopub.status.idle":"2021-08-03T17:07:33.900729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Creating sample id from graph number and edge endpoints","metadata":{}},{"cell_type":"code","source":"data <- data %>% mutate(Id = paste0(graph, \":\", from, \"-\", to))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:07:35.031582Z","iopub.execute_input":"2021-08-03T17:07:35.033188Z","iopub.status.idle":"2021-08-03T17:10:18.572398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Selecting only submission columns","metadata":{}},{"cell_type":"code","source":"data <- data %>% select(Id, Predicted = value)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:10:18.576135Z","iopub.execute_input":"2021-08-03T17:10:18.577745Z","iopub.status.idle":"2021-08-03T17:10:18.595581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T17:10:18.598365Z","iopub.execute_input":"2021-08-03T17:10:18.599853Z","iopub.status.idle":"2021-08-03T17:10:18.643687Z"},"trusted":true},"execution_count":null,"outputs":[]}]}