{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "library(data.table)\nlibrary(magrittr)\nlibrary(lubridate)\nlibrary(stringr)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Read only the columns we are going to use:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train <- fread(\"../input/train.csv\", select = c(\"user_location_city\", \"orig_destination_distance\",\n                                               \"srch_destination_id\", \"hotel_country\", \n                                               \"hotel_market\", \"is_booking\", \"hotel_cluster\"))"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Calculate hotel_cluster popularity:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "popular_hotel_cluster <- train[, table(hotel_cluster)]\npopular_hotel_cluster"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "tail(sort(popular_hotel_cluster), 5)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Calculate hotel_cluster popularity grouping by user_location_city and orig_destination_distance (in order to later use the data leak):"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "best_hotels_od_ulc <- train[!is.na(orig_destination_distance), .N, by = .(user_location_city, orig_destination_distance, hotel_cluster)]\nhead(best_hotels_od_ulc)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Calculate hotel_cluster popularity grouping by srch_destination_id, hotel_country and hotel_market. Each booking counts as [1]. \nEach click counts as [0.15]."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "best_hotels_search_dest <- train[, .(sum(is_booking + 0.15 * (!is_booking))), by = .(srch_destination_id, hotel_country, hotel_market, hotel_cluster)]\nhead(best_hotels_search_dest)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Remove training set and load test set:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# rm(train)\ngc()\n\ntest <- fread(\"../input/test.csv\", select = c(\"user_location_city\", \"orig_destination_distance\",\n                                               \"srch_destination_id\", \"hotel_country\", \n                                               \"hotel_market\"))\ntest[, id:=.I]"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "First of all, let's use the data leak.\nLet's create keys in the columns prior to merge:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "setkey(best_hotels_od_ulc, user_location_city, orig_destination_distance)\nsetkey(test, user_location_city, orig_destination_distance)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "new_test <- merge(test, best_hotels_od_ulc, )\nhead(new_test)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "The resulting data.table has more rown than test, due to duplicated (user_location_city,orig_destination_distance) pairs in the test set:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "new_test[, .(user_location_city,orig_destination_distance)][, duplicated(.SD)]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "new_test[1:100000, hotel_clusters := paste0(hotel_cluster, collapse = \",\"), by = .(id)]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "nrow(new_test)\nnrow(test)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Create submission name:"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "sub_name <- paste0(\"submission_\", now() %>% as.character() %>% str_replace_all(\"[ :]\", \"_\"), \".csv\")\nsub_name"
 }
],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"}}, "nbformat": 4, "nbformat_minor": 0}