{"cells":[{"metadata":{},"cell_type":"markdown","source":"Script aggregates the results from 2 models:\n* the [first model](https://www.kaggle.com/lennyom/iwildcam-2020-using-fastai-resnet50-mixup-and-tta)) was built on the original images resized to 128x128. I trained a ResNet50 using one cycle policy and mixup 10 epochs with a frozen body, and the next 10 epochs where all the model weights were retrained\n* the [second model](https://www.kaggle.com/lennyom/iwildcam2020-fastai-on-cropped-images) was also ResNet50 trained on cropped images, resized to 128x128, using one cycle policy 12 epochs with a frozen body, and the next 12 epochs where all weights were updated.\n\nPredictions are aggregated using the formula `0.3 * probabilities from model 1 + 0.7 * probabilities from model 2`"},{"metadata":{"trusted":true},"cell_type":"code","source":"library(dplyr)\nlibrary(readr)\nlibrary(purrr)\nlibrary(jsonlite)\nlibrary(lubridate)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_set_json <- fromJSON(\"../input/iwildcam-2020-fgvc7//iwildcam2020_test_information.json\")\n\n\nprob_croped <- read_csv(\"../input//iwildcam2020-model-output/predictions_croped_model.csv\")\nprob_full <- read_csv(\"../input//iwildcam2020-model-output/predictions_full_model.csv\")\n\nsubmission_org <- read_csv(\"../input//iwildcam-2020-fgvc7/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true},"cell_type":"code","source":"prob_croped <- prob_croped %>% mutate(file_names = stringr::str_split(prob_croped$Id, \"/\")  %>% map_chr(., 5),\n                        ID = gsub(\".jpg\", \"\", file_names),\n                        Id = NULL,\n                        file_names = NULL)\n\nprob_full <- prob_full %>% mutate(file_names = stringr::str_split(prob_full$Id, \"/\")  %>% map_chr(., 7),\n                                  ID = gsub(\".jpg\", \"\", file_names),\n                                  Id = NULL,\n                                  file_names = NULL)\n\ncats_cropped_img <- c(0, 2, 3, 4, 6, 7, 8, 9, 10, 12, 13, 14, 15, 16, 20, 24, 25, 26, 32, \n                           44, 50, 62, 67, 70, 71, 72, 73, 74, 77, 78, 80, 83, 86, 89, 90, 91, \n                           92, 94, 96, 97, 98, 99, 100, 101, 102, 103, 104, 106, 108, 110, 111, \n                           112, 113, 114, 115, 116, 118, 119, 120, 121, 122, 123, 124, 127, 129, \n                           130, 133, 134, 137, 139, 141, 142, 144, 145, 147, 150, 152, 153, 154, 156, \n                           159, 161, 162, 163, 166, 167, 170, 175, 177, 221, 227, 229, 230, 233, 234, \n                           235, 240, 242, 243, 245, 250, 252, 256, 257, 258, 259, 262, 265, 267, 268, \n                           273, 286, 291, 292, 294, 296, 299, 300, 301, 302, 306, 307, 309, 310, 315, \n                           316, 317, 318, 319, 320, 321, 322, 323, 324, 325, 326, 327, 328, 330, 332, \n                           333, 334, 335, 336, 337, 338, 339, 340, 341, 342, 344, 345, 346, 347, 349, \n                           350, 352, 353, 354, 355, 356, 357, 370, 371, 372, 374, 375, 376, 377, 378, \n                           379, 380, 382, 384, 385, 389, 390, 391, 402, 404, 405, 406, 407, 408, 409, \n                           410, 412, 413, 414, 415, 416, 417, 418, 419, 422, 454, 558, 559, 561, 562, \n                           563, 564, 565, 566, 567, 568, 569, 570, 571)\n\ncats_full_img <- c(0, 2, 3, 4, 6, 7, 8, 9, 10, 12, 13, 14, 15, 16, 20, 24, 25, 26, 32,\n               44, 50, 62, 67, 70, 71, 72, 73, 74, 77, 78, 79, 80, 83, 86, 89, 90,\n               91, 92, 94, 96, 97, 98, 99, 100, 101, 102, 103, 104, 106, 108, 110,\n               111, 112, 113, 114, 115, 116, 118, 119, 120, 121, 122, 123, 124, 127,\n               129, 130, 133, 134, 137, 139, 141, 142, 144, 145, 147, 150, 152, 153,\n               154, 156, 159, 161, 162, 163, 166, 167, 170, 175, 177, 198, 221, 227,\n               229, 230, 233, 234, 235, 240, 242, 243, 245, 250, 251, 252, 253, 256,\n               257, 258, 259, 262, 265, 267, 268, 273, 286, 290, 291, 292, 294, 296,\n               299, 300, 301, 302, 306, 307, 309, 310, 315, 316, 317, 318, 319, 320,\n               321, 322, 323, 324, 325, 326, 327, 328, 330, 332, 333, 334, 335, 336,\n               337, 338, 339, 340, 341, 342, 344, 345, 346, 347, 348, 349, 350, 352,\n               353, 354, 355, 356, 357, 370, 371, 372, 374, 375, 376, 377, 378, 379,\n               380, 382, 384, 385, 389, 390, 391, 402, 404, 405, 406, 407, 408, 409,\n               410, 412, 413, 414, 415, 416, 417, 418, 419, 420, 422, 454, 558, 559,\n               561, 562, 563, 564, 565, 566, 567, 568, 569, 570, 571)\n\ncolnames(prob_croped) <- c(cats_cropped_img, \"ID\")\n\ncats_full_img[!cats_full_img %in% cats_cropped_img]\n\nprob_croped <- prob_croped %>% mutate(\n  \"79\" = 0,\n  \"198\" = 0,\n  \"251\" = 0,\n  \"253\" = 0,\n  \"290\" = 0,\n  \"348\" = 0,\n  \"420\" = 0\n)\n\nprob_croped <- prob_croped %>%\n  relocate(as.character(cats_full_img))\n\ncolnames(prob_full) <- c(cats_full_img, \"ID\")\n\nnames(prob_croped) == names(prob_full)\n\ncroped_ext <- tibble(ID = prob_full$ID) %>% left_join(prob_croped)\n\ncroped_ext[is.na(croped_ext)] <- 0\n\nnew_probs <- croped_ext[,2:ncol(croped_ext)] * 0.7 + prob_full[,1:(ncol(croped_ext) - 1)] * 0.3\n\nargmax_predictions <- tibble(index = new_probs %>% apply(1, which.max))\n\ndf_map_cat <- tibble(index = 1:216, cats_full_img)\n\ntrue_pred_cat <- argmax_predictions %>% left_join(df_map_cat)\n\npredictions_with_ids <- tibble(Id = prob_full$ID, Category = true_pred_cat$cats_full_img)\n\nsubmission <- submission_org %>% \n  mutate(Category = NULL) %>%\n  left_join(predictions_with_ids)\n\ntest_images <- test_set_json[\"images\"] %>% as.data.frame() %>%\n  mutate(datetime = as_datetime(images.datetime)) %>%\n  group_by(images.location) %>%\n  arrange(datetime) %>%\n  mutate(lag_datetime = dplyr::lag(datetime, n = 1, default = NA),\n         diff_time = difftime(datetime, lag_datetime, units = \"secs\"))\n\ntest_images$new_seq_id <- 0\nfor (i in 2:nrow(test_images)){\n  current_seq_id <-  test_images$new_seq_id[i-1] \n  test_images$new_seq_id[i] <- ifelse(\n    test_images$images.location[i-1] != test_images$images.location[i] | test_images$diff_time[i] > 20, \n    current_seq_id + 1, current_seq_id)\n}\n\nsubmission_with_seq_id <- submission %>%\n  left_join(test_images %>% ungroup() %>% select(images.id, new_seq_id), \n            by = c(\"Id\" = \"images.id\"))\n\nmost_common_category_with_empty <- submission_with_seq_id %>%\n  group_by(new_seq_id) %>% \n  count(new_seq_id, Category) %>%\n  slice(which.max(n))\n\nmost_common_category_without_empty <- submission_with_seq_id %>%\n  filter(Category != 0) %>% \n  group_by(new_seq_id) %>% \n  count(new_seq_id, Category) %>%\n  slice(which.max(n))\n\nseq_id_predicted_category <- most_common_category_with_empty %>% \n  left_join(most_common_category_without_empty, by = \"new_seq_id\") %>%\n  select(new_seq_id, Category.x, Category.y) %>%\n  mutate(final_category = ifelse(is.na(Category.y), Category.x, Category.y)) %>%\n  select(new_seq_id, final_category)\n\nfinal_pred_with_ids <- submission_with_seq_id %>% \n  left_join(seq_id_predicted_category)\n\nsubmission_corrected_categories <- final_pred_with_ids %>% \n  mutate(Category = NULL,\n         new_seq_id = NULL) %>%\n  rename(Category = final_category)\n\n                                               \n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"write_csv(submission_corrected_categories, \"submission_fixed_seq.csv\")  ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}