{"cells":[{"metadata":{"_uuid":"91fa64373b972fd9ee594f7c021ec2857d5e7326","_execution_state":"idle","trusted":true},"cell_type":"code","source":"## Importing packages\n\n# This R environment comes with all of CRAN and many other helpful packages preinstalled.\n# You can see which packages are installed by checking out the kaggle/rstats docker image: \n# https://github.com/kaggle/docker-rstats\n\nlibrary(tidyverse)# metapackage with lots of helpful functions\nlibrary(readr)\nlibrary(xgboost)\n\n\n\n\n## Running code\n\n# In a notebook, you can run a single code cell by clicking in the cell and then hitting \n# the blue arrow to the left, or by clicking in the cell and pressing Shift+Enter. In a script, \n# you can run code by highlighting the code you want to run and then clicking the blue arrow\n# at the bottom of this window.\n\n## Reading in files\n\n# You can access files from datasets you've added to this kernel in the \"../input/\" directory.\n# You can see the files added to this kernel by running the code below. \n\nlist.files(path = \"../input\")\ntr <- read_csv(\"../input/yelp-restaurant-photo-classification/train.csv\")\nlabels <- strsplit(tr$labels,\" \")\ndim(tr)\nlabels <- sapply(labels, as.integer)\n\nlabels <- lapply(labels, function(x) 0:8 %in% x)\ndim(labels)\n                 \n\n\n\n#colnames(labels) <- 0:8\n#rownames(labels) <- tr$business_id\n\n\np2b.tr <- read_csv(\"../input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv\")\np2b.te <- read_csv(\"../input/yelp-restaurant-photo-classification/test_photo_to_biz.csv\")\n                 \nfeat.tr <- read_csv(\"../input/extracted-features-for-yelp/feat03-21k.tr.csv\")\nfeat.te <- read_csv(\"../input/extracted-features-for-yelp/feat03-21k.te.csv\")\n\ngen.bfeat <- function(p2b, dat) {\n  ddply(p2b, .(business_id), function(df) {\n    df <- merge(df, dat)[,-(1:2)]\n    ret <- llply(df, function(v) {\n      c(mean=mean(v),\n        sd=sd(v),\n        quantile(v, c(0.6,0.8,0.9,0.95,1)))\n    })\n    names(ret) <- colnames(df)\n    unlist(ret)\n  }, .progress=\"tk\")\n}                 \n\ncol.sd <- sapply(feat.tr[,-1], sd)\nbfeat.tr <- gen.bfeat05(p2b.tr, feat.tr)\nbfeat.te <- gen.bfeat05(p2b.te, feat.te)    \n                 \nsel.feat <- names(col.sd)[col.sd>0.007]\nsel.col <- colnames(bfeat.tr)[sub(\"\\\\..*\", \"\", colnames(bfeat.tr)) %in% sel.feat]\nx <- as.matrix(bfeat.tr[,sel.col])\nx.te <- as.matrix(bfeat.te[,sel.col])\n                 \nfits <- llply(0:8, function(i) {\n  target <- paste0(\"label\", i)\n\n  x2 <- cbind(x, labels[,(0:8)<i])\n  dtrain <- xgb.DMatrix(x2, label=tr[[target]], missing=NA)\n\n  params <- list(objective=\"binary:logistic\", eval_metric=\"logloss\", eta=0.01, max_depth=10, subsample=0.5, colsample_bytree=0.3)\n  xgb.train(params, dtrain, nrounds=500)\n}, .progress=\"win\")\n                 \nmat.subm <- matrix(0, nrow=nrow(x.te), ncol=9)\ncolnames(mat.subm) <- 0:8\nfor (i in 0:8) {\n  target <- paste0(\"label\", i)\n\n  x2.te <- cbind(x.te, mat.subm[,(0:8)<i]>0.4)\n  dtest <- xgb.DMatrix(x2.te, missing=NA)\n  mat.subm[,i+1] <- predict(fits[[i+1]], dtest)\n}\n\npred <- mat.subm>0.4\ncolnames(pred) <- 0:8\n\nsubm <- data.frame(business_id=sort(unique(p2b.te$business_id)), labels=laply(1:nrow(pred), function(i) paste(which(pred[i,])-1, collapse=\" \")))\nwrite.csv(subm, gzfile(\"submmission.csv.gz\"), row.names=F)\n                 \n## Saving data\n\n# If you save any files or images, these will be put in the \"output\" directory. You \n# can see the output directory by committing and running your kernel (using the \n# Commit & Run button) and then checking out the compiled version of your kernel.","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}