{"cells":[{"metadata":{"_uuid":"710ba5adf878af1f3f5e22856bd6f3311a397063","_execution_state":"idle","trusted":true},"cell_type":"code","source":"## Importing packages\n\n# This R environment comes with all of CRAN and many other helpful packages preinstalled.\n# You can see which packages are installed by checking out the kaggle/rstats docker image: \n# https://github.com/kaggle/docker-rstats\n\nlibrary(tidyverse) # metapackage with lots of helpful functions\n\n## Running code\n\n# In a notebook, you can run a single code cell by clicking in the cell and then hitting \n# the blue arrow to the left, or by clicking in the cell and pressing Shift+Enter. In a script, \n# you can run code by highlighting the code you want to run and then clicking the blue arrow\n# at the bottom of this window.\n\n## Reading in files\n\n# You can access files from datasets you've added to this kernel in the \"../input/\" directory.\n# You can see the files added to this kernel by running the code below. \n\nlist.files(path = \"../input\")\n\n## Saving data\n\n# If you save any files or images, these will be put in the \"output\" directory. You \n# can see the output directory by committing and running your kernel (using the \n# Commit & Run button) and then checking out the compiled version of your kernel.","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"fbfabedf51314d52f3e25b972c286038684c820a"},"cell_type":"code","source":"library(caret)\n# text mining libraries\nlibrary(tidytext)\nlibrary(SnowballC)\nlibrary(tm)\nlibrary(ggplot2)\nlibrary(RColorBrewer)\nlibrary(wordcloud)\nlibrary(topicmodels)\nlibrary(data.table)\nlibrary(stringi)\nlibrary(qdap)\nlibrary(dplyr)\nlibrary(rJava)\nlibrary(syuzhet)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a589ceb8e1779cfb0cd2acc5766a20ad5b23d5a"},"cell_type":"code","source":"train_questions <- read_csv('../input/train.csv',n_max=11274)\ndim(train_questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c5073d9db97c38b692844b6226a3c0add489411"},"cell_type":"code","source":"train_questions$target <- as.factor(train_questions$target)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"816dfc4dccedcdab773f86c640f9f60783591edf"},"cell_type":"code","source":"str(train_questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe3b792500ad00e14eae95b8b5c474ca8a45988d"},"cell_type":"code","source":"# Create document corpus with question text and clean up\ntrainCorpus<- VCorpus(VectorSource(train_questions$question_text)) \nwriteLines(strwrap(trainCorpus[[787]]$content,60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0163a3b97186e0429a5e1b4af3a613504875c3fc"},"cell_type":"code","source":"# Replace Contractions e.g. shouldn't -> should not\n#trainCorpus <- tm_map(trainCorpus,content_transformer(replace_contraction))\n# convert to Lowercase\ntrainCorpus <- tm_map(trainCorpus,content_transformer(tolower))\n# Remove the links (URLs)\nremoveURL <- function(x) gsub(\"http[^[:space:]]*\", \"\", x)  \ntrainCorpus <- tm_map(trainCorpus, content_transformer(removeURL))\n# Remove anything except the english words and space\nremoveNumPunct <- function(x) gsub(\"[^[:alpha:][:space:]]*\", \"\", x)   \ntrainCorpus <- tm_map(trainCorpus, content_transformer(removeNumPunct))\n# Remove Punctuations\ntrainCorpus <- tm_map(trainCorpus,removePunctuation)\n# Remove Extra Whitespaces\ntrainCorpus <- tm_map(trainCorpus,stripWhitespace)\n# Remove Stopwords\nmyStopWords<- c((stopwords(\"en\")))\ntrainCorpus<- tm_map(trainCorpus,removeWords, myStopWords)\n# Remove Single letter words\nremoveSingle <- function(x) gsub(\" . \", \" \", x) \ntrainCorpus <- tm_map(trainCorpus, content_transformer(removeSingle))\n# display a random text for validation\nwriteLines(strwrap(trainCorpus[[787]]$content,60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6468a6113768b1644398b0c07967a39e87dbe892"},"cell_type":"code","source":"test_questions <- read_csv('../input/test.csv')#,n_max=10000)\ndim(test_questions)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"6b2e02774b6da6776c684f2b4e12d605ab317ea7"},"cell_type":"code","source":"str(test_questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b3a2974e21361c13564ee88338f7011ea2f5db9"},"cell_type":"code","source":"# Create document corpus with question text and clean up\ntestCorpus<- VCorpus(VectorSource(test_questions$question_text)) \nwriteLines(strwrap(testCorpus[[787]]$content,60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa4da7a8086c871cb7bccb894785092dca1da208"},"cell_type":"code","source":"# Replace Contractions e.g. shouldn't -> should not\n#testCorpus <- tm_map(testCorpus,content_transformer(replace_contraction))\n# convert to Lowercase\ntestCorpus <- tm_map(testCorpus,content_transformer(tolower))\n# Remove the links (URLs)\nremoveURL <- function(x) gsub(\"http[^[:space:]]*\", \"\", x)  \ntestCorpus <- tm_map(testCorpus, content_transformer(removeURL))\n# Remove anything except the english words and space\nremoveNumPunct <- function(x) gsub(\"[^[:alpha:][:space:]]*\", \"\", x)   \ntestCorpus <- tm_map(testCorpus, content_transformer(removeNumPunct))\n# Remove Punctuations\ntestCorpus <- tm_map(testCorpus,removePunctuation)\n# Remove Extra Whitespaces\ntestCorpus <- tm_map(testCorpus,stripWhitespace)\n# Remove Stopwords\nmyStopWords<- c((stopwords(\"en\")))\ntestCorpus<- tm_map(testCorpus,removeWords, myStopWords)\n# Remove Single letter words\nremoveSingle <- function(x) gsub(\" . \", \" \", x) \ntestCorpus <- tm_map(testCorpus, content_transformer(removeSingle))\n# display a random text for validation\nwriteLines(strwrap(testCorpus[[787]]$content,60))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bca08da84b4b103d5732e4ca1d4b506fac79768a"},"cell_type":"code","source":"train.dtm <- DocumentTermMatrix(trainCorpus, control=list(wordLengths=c(1,Inf)))\ntrain.dtm2 <- removeSparseTerms(train.dtm, sparse = 0.99)\ntest.dtm <- DocumentTermMatrix(testCorpus, control=list(wordLengths=c(1,Inf)))\ntest.dtm2 <- removeSparseTerms(test.dtm, sparse = 0.99)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"686e9d3d3f41573bc7f3bf31d2011f3929148aef"},"cell_type":"code","source":"train.df <- as.data.frame(as.matrix(train.dtm2[,intersect(colnames(train.dtm2), colnames(test.dtm2))]))\n#train.df <- as.data.frame(as.matrix(train.dtm2))\ntrain.df$qid <- train_questions$qid\ntrain.df$target <- train_questions$target\ndim(train.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e20ab021d4f0593239da3ab0f138270153a86262"},"cell_type":"code","source":"test.df <- as.data.frame(as.matrix(test.dtm2[,intersect(colnames(test.dtm2), colnames(train.dtm2))]))\n#test.df <- as.data.frame(as.matrix(test.dtm2))\ntest.df$qid <- test_questions$qid\ndim(test.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce871d46c4d035fdf367ef8c50cd1fc69bab58bf"},"cell_type":"code","source":"head(train.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8e8ad63af4a2d5a4455d5fc6c38ba7702babc63"},"cell_type":"code","source":"head(test.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c3146bc1a6b4d73f13b5fe47653c4b6ae0b8322"},"cell_type":"code","source":"#library(tidytext)\n#train.df <- tidy(train.dtm)\n#dim(train.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a5f4e22511b0ec5dc92a2fe1d063539712f3479"},"cell_type":"code","source":"df.train <- train.df\ndf.test <- train.df\nlibrary(kernlab)\ndf.model<-ksvm(target~., data= df.train, kernel=\"rbfdot\")\nsummary(df.model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c2c05bf977a9f95f79a251e800a394d3917c9c7"},"cell_type":"code","source":"df.pred<-predict(df.model, df.test)\ncon.matrix<-confusionMatrix(df.pred, df.test$target)\nprint(con.matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29c0cfd87bb6fab9f4c2147694c89ffe25c8a0e9"},"cell_type":"code","source":"df.test <- test.df[c(1:11274),]\ndf.pred <- predict(df.model, df.test)\ndf.test$target <- df.pred\nsubmission1 <- df.test[,c(\"qid\",\"target\")]\ncolnames(submission1)[colnames(submission1)==\"target\"] <- \"prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48428ca4835aa6933b619f4f4144fb49cd886895"},"cell_type":"code","source":"df.test <- test.df[c(11275:22548),]\ndf.pred <- predict(df.model, df.test)\ndf.test$target <- df.pred\nsubmission2 <- df.test[,c(\"qid\",\"target\")]\ncolnames(submission2)[colnames(submission2)==\"target\"] <- \"prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e55a7d589efb1c000e06fdae897e95f7bca54be"},"cell_type":"code","source":"df.test <- test.df[c(22549:33822),]\ndf.pred <- predict(df.model, df.test)\ndf.test$target <- df.pred\nsubmission3 <- df.test[,c(\"qid\",\"target\")]\ncolnames(submission3)[colnames(submission3)==\"target\"] <- \"prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6aeb4fb55031a02ef361bca9c922233c9124dee4"},"cell_type":"code","source":"df.test <- test.df[c(33823:45096),]\ndf.pred <- predict(df.model, df.test)\ndf.test$target <- df.pred\nsubmission4 <- df.test[,c(\"qid\",\"target\")]\ncolnames(submission4)[colnames(submission4)==\"target\"] <- \"prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00cb9807440f765d5574f02a5c2a27cbac6eceab"},"cell_type":"code","source":"df.test <- test.df[c(45097:56370),]\ndf.pred <- predict(df.model, df.test)\ndf.test$target <- df.pred\nsubmission5 <- df.test[,c(\"qid\",\"target\")]\ncolnames(submission5)[colnames(submission5)==\"target\"] <- \"prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bb9b4f9a9c81cf8552140287fafc27ec3ba4ad5"},"cell_type":"code","source":"submission <- rbind(submission1,submission2,submission3,submission4,submission5)\nwrite.csv(submission, \"submission.csv\",row.names = FALSE)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}