{"cells":[{"metadata":{"_uuid":"fc0b43743f5b28e8feb256057f093bd637731bbb","_execution_state":"idle","trusted":true},"cell_type":"code","source":"## Importing packages\n\n# This R environment comes with all of CRAN and many other helpful packages preinstalled.\n# You can see which packages are installed by checking out the kaggle/rstats docker image: \n# https://github.com/kaggle/docker-rstats\n\nlibrary(tidyverse) # metapackage with lots of helpful functions\n\n## Running code\n\n# In a notebook, you can run a single code cell by clicking in the cell and then hitting \n# the blue arrow to the left, or by clicking in the cell and pressing Shift+Enter. In a script, \n# you can run code by highlighting the code you want to run and then clicking the blue arrow\n# at the bottom of this window.\n\n## Reading in files\n\n# You can access files from datasets you've added to this kernel in the \"../input/\" directory.\n# You can see the files added to this kernel by running the code below. \n\nlist.files(path = \"../input\")\n\n## Saving data\n\n# If you save any files or images, these will be put in the \"output\" directory. You \n# can see the output directory by committing and running your kernel (using the \n# Commit & Run button) and then checking out the compiled version of your kernel.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84e6af4b44193ec43d11af77547f0c9295daeedc"},"cell_type":"code","source":"# A quick and dirty lightgbm run. Still need to do parameter tuning and cv\n\nlibrary(data.table)\nlibrary(lightgbm)\n\n# read data\ndtrain <- fread('../input/train.csv', drop = 'MachineIdentifier')\ndtest <- fread('../input/test.csv')\n\n# check size in memory\nprint(object.size(dtrain), units='Gb')\nprint(object.size(dtest), units='Gb')\n\n#sample 1/2 of training data. Will optimize later so we can use all of the data\nrows = sample(1:nrow(dtrain), size = nrow(dtrain)/2, replace=FALSE)\n\n#set dtrain to only sample rows\ndtrain <- dtrain[rows, ]\nprint(object.size(dtrain), units='Gb')\n\n#assign target variable to y_train\ny_train <- dtrain$HasDetections\ndtrain[, HasDetections := NULL]\n\n# assign test data ids to id_test\nid_test <- dtest$MachineIdentifier\ndtest[, MachineIdentifier := NULL]\n\n# save # rows in dtrain and dtest\nnrow_dtrain <- nrow(dtrain)\nnrow_dtest <- nrow(dtest)\n\n# combine dtrain and dtest for preprocessing\nalldata <- rbindlist(list(dtrain, dtest))\n\n# remove dtrain and dtest\nrm(dtrain, dtest)\ngc()\n\n# get vector of character columns\nchar_cols <- names(which(sapply(alldata, class) == 'character'))\n\n# convert character columns to integer\nalldata[, (char_cols) := lapply(.SD, function(x) as.integer(as.factor(x))), .SDcols = char_cols]\n\n# split back into dtrain and dtest\ndtrain <- alldata[1:nrow_dtrain, ]\ndtest <- alldata[(nrow_dtrain+1):nrow(alldata)]\n\nrm(alldata); gc();\n\n# Set up processed data for lgbm\nx_train <- lgb.Dataset(data = data.matrix(dtrain), label = y_train)\nx_test <- data.matrix(dtest)\n\nrm(dtrain, dtest); gc();\n\nparams = list(\n    boosting_type = 'gbdt', \n    objective = 'binary',\n    metric = 'auc', \n    nthread = 4, \n    learning_rate = 0.05, \n    max_depth = 5,\n    num_leaves = 40,\n    sub_feature = 0.7, \n    sub_row = 0.7, \n    bagging_freq = 1,\n    lambda_l1 = 0.1, \n    lambda_l2 = 0.1\n)\n\n# train model\nlgbm_mod <- lgb.train(params = params,\n                      nrounds = 1000,\n                      data = x_train, \n                      verbose = 1\n    )\n\n# make predictions\npreds <- predict(lgbm_mod, data = x_test)\n\n# set up submission frame\nsub <- data.table(\n    MachineIdentifier = id_test,\n    HasDetections = preds\n)\n\n# write to disk\nfwrite(sub, file = 'submission_lgbm.csv', row.names = FALSE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f69cc5c0035b16051c61a6eb2409008d74104924"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}