{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "e8023446-3eab-5487-b215-186c8942b310"
      },
      "source": [
        "This is a learning exercise for me where I would like to get suggestions from you folks! I am forced to use just a small proportion of the train data since R is quite slow in processing !\n",
        "\n",
        "Please do share your comments !"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b8384683-56d8-d22e-5afd-cad4249c6c4e"
      },
      "outputs": [],
      "source": [
        "# LIST OF PACKAGES TO BE USED\n",
        "library(dplyr)\n",
        "library(data.table)\n",
        "library(dtplyr)\n",
        "library(topicmodels)\n",
        "library(tidytext)\n",
        "library(ggplot2)\n",
        "library(randomForest)\n",
        "library(tm)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "245fb440-dfdb-af6d-0559-c41743d92f44"
      },
      "outputs": [],
      "source": [
        "df <- fread(\"../input/train.csv\")\n",
        "# just working on first 100k obs, due to processing limitations\n",
        "df <- df[1:100000,]\n",
        "print(dim(df))\n",
        "head(df, 3)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7b6b9f67-f8cf-60c3-ae22-ee8a09e6fa69"
      },
      "outputs": [],
      "source": [
        "# lowercase conversion, removing html/http/image links, and then the STOPWORDS \n",
        "cleanup <- function(x){\n",
        "  x <- tolower(x)\n",
        "  x <- gsub(\"<img src.*?>\", \"\", x)\n",
        "  x <- gsub(\"http\\\\S+\", \"\", x)\n",
        "  x <- gsub(\"\\\\[math\\\\]\", \"\", x)    # text between [] refers to tags e.g. [math]\n",
        "  x <- gsub(\"<.*?>\", \"\", x)\n",
        "  x <- gsub(\"\\n\", \" \", x)                 # replace newline with a space\n",
        "  x <- gsub(\"\\\\s+\", \" \", x)                # multiple spaces into one\n",
        "  # using tm_map to remove stopwords\n",
        "  docs <- Corpus(VectorSource(x))\n",
        "  docs <- tm_map(docs, removeWords, stopwords('en'))\n",
        "  docs <- tm_map(docs, removePunctuation)    # dont remove punct so early in the analysis\n",
        "  docs <- tm_map(docs, stripWhitespace)\n",
        "  xxx <- sapply(docs, function(i) i)\n",
        "  data_content <- data.frame(text = xxx, stringsAsFactors = FALSE)\n",
        "  return(data_content$text)\n",
        "}\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b7608b3d-b275-720a-448e-34156b5a7148"
      },
      "outputs": [],
      "source": [
        "df$question1 <- cleanup(df$question1)\n",
        "df$question2 <- cleanup(df$question2)\n",
        "head(df,3)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "22f540dc-6ac7-ae92-99c5-4f4a01e6d080"
      },
      "outputs": [],
      "source": [
        "# below code shall create tokens(words) per question - \n",
        "# these can be used later on to calculate tfidf weights\n",
        "\n",
        "#######################################################################\n",
        "### baseline model  ####\n",
        "### tf-idf weight as weigt of common words\n",
        "### common words\n",
        "### diff between nchar of q2, q1\n",
        "######################################################################\n",
        "### for question1\n",
        "tokens_q1 <- df %>%\n",
        "  unnest_tokens(word, question1, drop = FALSE, token = \"regex\", pattern = \" \") %>%\n",
        "  count(id, word) %>%\n",
        "  ungroup()\n",
        "tokens_q1 <- tokens_q1[df, on = \"id\"]\n",
        "colnames(tokens_q1)[1:3] <- c(\"id1\", \"word1\", \"n1\")\n",
        "tokens_q1 <- tokens_q1[,c(\"id1\", \"question1\", \"word1\", \"n1\"),with = FALSE]\n",
        "# calculate tf-idf weights\n",
        "tf.idf1 <- tokens_q1 %>% bind_tf_idf(word1, question1, n1) %>%\n",
        "                                select(id1, question1, word1, tf, idf, tf_idf)\n",
        "\n",
        "###  for question2\n",
        "tokens_q2 <- df %>%\n",
        "  unnest_tokens(word, question2, drop = FALSE, token = \"regex\", pattern = \" \") %>%\n",
        "  count(id, word) %>%\n",
        "  ungroup()\n",
        "tokens_q2 <- tokens_q2[df, on = \"id\"]\n",
        "colnames(tokens_q2)[1:3] <- c(\"id2\", \"word2\", \"n2\")\n",
        "tokens_q2 <- tokens_q2[,c(\"id2\", \"question2\", \"word2\", \"n2\"),with=FALSE]\n",
        "# calculate tf-idf weights\n",
        "tf.idf2 <- tokens_q2 %>% bind_tf_idf(word2, question2, n2) %>%\n",
        "                         select(id2, question2, word2, tf, idf, tf_idf)\n",
        "\n",
        "head(tf.idf2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ee8d6f22-1c3d-4f6d-2dff-c5a7bc9d7f8a"
      },
      "outputs": [],
      "source": [
        "##  as a baseline model - all that i'm doing here is get the ratio of common words in \n",
        "##  both questions to the total number of unique words\n",
        "## using the tf-idf weights to give exposure to common words\n",
        "\n",
        "func <- function(x){\n",
        "  id.check <- x$id1[1] == tf.idf2$id2    # boolean vector to subset the question2 for same id \n",
        "  \n",
        "  words1 <- x$word1\n",
        "  words2 <- tf.idf2$word2[id.check]\n",
        "  common <- intersect(words1, words2)    # list of common words in both\n",
        "  uncommon.q1 <- setdiff(words1, words2) # words not present in question1\n",
        "  uncommon.q2 <- setdiff(words2, words1) # words not present in question2\n",
        "  len_common_words <- length(common)\n",
        "  \n",
        "  len_q1 <- nchar(x$question1[1])\n",
        "  len_q2 <- nchar(tf.idf2$question2[id.check][1])\n",
        "  diff_len <- abs(len_q1 - len_q2)       # difference in length of characters\n",
        "  \n",
        "  tfidf.wt1 <- x$tf_idf\n",
        "  tfidf.wt2 <- tf.idf2$tf_idf[id.check]\n",
        "  # calculate how similar both questions are based on tfidf weights\n",
        "  # positive effect for the common words and negative exposure for the uncommon words\n",
        "  w1_shared_wts <- tfidf.wt1[match(common, words1)]\n",
        "  w1_unshared_wts <- tfidf.wt1[match(uncommon.q1, words1)]\n",
        "  w2_shared_wts <- tfidf.wt2[match(common, words2)]\n",
        "  w2_unshared_wts <- tfidf.wt2[match(uncommon.q2, words2)]\n",
        "  ratio_commonality = (sum(c(w1_shared_wts,w2_shared_wts))-sum(c(w1_unshared_wts,w2_unshared_wts)))/(sum(tfidf.wt1, tfidf.wt2))\n",
        "  return(list(len_common_words, ratio_commonality, diff_len))\n",
        "}\n",
        "\n",
        "ans = tf.idf1[ , c(\"len_common_words\", \"ratio_commonality\", \"diff_len\") := func(.SD) , keyby = id1, .SDcols = c(colnames(tf.idf1))]\n",
        "ans <- ans[, c(\"id1\", \"len_common_words\", \"ratio_commonality\", \"diff_len\"), with = FALSE]\n",
        "colnames(ans)[1] = \"id\"\n",
        "ans <- ans[!duplicated(ans$id),]\n",
        "ans <- df[ans, on = \"id\"]\n",
        "ans$is_duplicate <- factor(ans$is_duplicate)\n",
        "ans$ratio_commonality[is.na(ans$ratio_commonality)] <- min(ans$ratio_commonality, na.rm = TRUE)\n",
        "\n",
        "head(ans)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b1c1abaf-83eb-b16e-291c-efef0d237855"
      },
      "outputs": [],
      "source": [
        "#####################################################################################\n",
        "##    first model using 3 features : ratio_commonality, len_commn_words, diff_len  ##\n",
        "#####################################################################################\n",
        "# split into train-test\n",
        "set.seed(1000)\n",
        "s <- sample(1:nrow(ans), 0.7*nrow(ans), replace = FALSE)\n",
        "train <- ans[s,]\n",
        "test <- ans[-s,]\n",
        "\n",
        "\n",
        "rf <- randomForest(is_duplicate ~ len_common_words + ratio_commonality+\n",
        "                     diff_len, data = train, ntree = 1000, mtry = 3)\n",
        "pred <- predict(rf, newdata = test)\n",
        "\n",
        "# accuracy check\n",
        "sum(pred==test$is_duplicate)/nrow(test)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3a950a68-8411-b6ef-edd9-3336d71c5ab4"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}