{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "65201a94-efc0-9794-fdea-5b2d340a279e"
      },
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24d6611a-e7e6-fdb1-6d06-7abb9d6cefdf"
      },
      "outputs": [],
      "source": [
        "from nltk.corpus import stopwords\n",
        "import pandas as pd\n",
        "import numpy as np\n",
        "from sklearn.metrics import log_loss\n",
        "from scipy.optimize import minimize\n",
        "\n",
        "stops = set(stopwords.words(\"english\"))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cd8bc8dd-170a-f5ee-836f-af64a99f7f56"
      },
      "outputs": [],
      "source": [
        "def word_match_share(row):\n",
        "    q1words = {}\n",
        "    q2words = {}\n",
        "    for word in str(row['question1']).lower().split():\n",
        "        if word not in stops:\n",
        "            q1words[word] = 1\n",
        "    for word in str(row['question2']).lower().split():\n",
        "        if word not in stops:\n",
        "            q2words[word] = 1\n",
        "    if len(q1words) == 0 or len(q2words) == 0:\n",
        "        # The computer-generated chaff includes a few questions that are nothing but stopwords\n",
        "        return 0\n",
        "    shared_words_in_q1 = [w for w in q1words.keys() if w in q2words]\n",
        "    shared_words_in_q2 = [w for w in q2words.keys() if w in q1words]\n",
        "    R = (len(shared_words_in_q1) + len(shared_words_in_q2))/(len(q1words) + len(q2words))\n",
        "    return R\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c69e5fba-9d99-37a4-100e-ff1d171cd6fb"
      },
      "outputs": [],
      "source": [
        "#LOAD TRAINSET, TESTSET\n",
        "train = pd.read_csv(\"../input/train.csv\")\n",
        "train[ 'R' ] = train.apply( word_match_share, axis=1, raw=True )\n",
        "print( train.head() ) \n"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "cba0d536-e7c0-b63a-4736-fc7ff5cf7571"
      },
      "source": [
        "So as it is apparent from the train data, only those which have exactly same meaning are considered duplicates.\n",
        "however since in the test data, our task is to decide the predict the probability that the questions are duplicates, it gets easier than this\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d5d88969-86b4-b18b-ddc6-8e253083ddb6"
      },
      "outputs": [],
      "source": [
        "test = pd.read_csv(\"../input/test.csv\", index_col=False )\n",
        "test['R'] = test.apply( word_match_share, axis=1, raw=True )\n",
        "print( test.head() )\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9c97499e-581d-afee-3db3-e3d186bc6255"
      },
      "outputs": [],
      "source": [
        "#Mean target\n",
        "GLOBAL_MEAN = np.mean( train['is_duplicate'] ) \n",
        "print( 'Mean is_duplicated', GLOBAL_MEAN )\n",
        "\n",
        "#OPTIMIZE FUNCTIONS\n",
        "def minimize_train_log_loss( W ):\n",
        "    train[\"prediction\"] = GLOBAL_MEAN + train[\"R\"] * W[0] + W[1]\n",
        "    score = log_loss( train['is_duplicate'], train['prediction'] )\n",
        "    print(  score , W )\n",
        "    return( score )\n",
        "\n",
        "res = minimize(minimize_train_log_loss, [0.00,  0.00], method='Nelder-Mead', tol=1e-4, options={'maxiter': 400})\n",
        "W = res.x\n",
        "print( 'Best weights: ',W )\n"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "c7ee862a-5901-357d-6189-70ce6ef991c6"
      },
      "source": [
        "For this trial run, we we will create two series for each question in each row and find the number of  .\n",
        "similar words in them.\n",
        "We find the probability by dividing this by the length of each series.\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "82c0127f-1d7b-6d9f-26f6-6cb1982c0156"
      },
      "outputs": [],
      "source": [
        "\n",
        "#APPLY TO TESTSET\n",
        "test[\"is_duplicate\"] = test[\"R\"] \n",
        "test[ ['test_id','is_duplicate'] ].to_csv(\"count_words_benchmark.csv\", header=True, index=False)\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c636b090-e534-bc78-df4a-39bb4f765d8f"
      },
      "outputs": [],
      "source": [
        "test.shape[0]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9622a27a-c1b3-0d4e-cf16-dbb0392255b4"
      },
      "outputs": [],
      "source": [
        "def prob_sim(rows):\n",
        "        q1=pd.Series(get_words(rows['question1']))\n",
        "        q2=pd.Series(get_words(rows['question2']))\n",
        "        n1=len([w for w in q1.values if w in q2.values])\n",
        "        n2=len([w for w in q2.values if w in q1.values])\n",
        "        n=len(q1)+len(q2)\n",
        "        return (n1+n2)/n\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bc8bf170-9ea3-1fde-eb39-faf1d59b5135"
      },
      "outputs": [],
      "source": [
        "test['prob']=test.apply(prob_sim,axis=1,raw=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "563eb94a-241a-fdeb-68d5-889a1da7c34a"
      },
      "outputs": [],
      "source": [
        "test_dummy['prob']=test_dummy.apply(prob_sim,axis=1,raw=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9f042aa2-f327-0113-c736-d1003bb40e2c"
      },
      "outputs": [],
      "source": [
        "test_dummy['prob']"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "87faaa21-9109-2b63-adeb-1681ec9d0995"
      },
      "outputs": [],
      "source": [
        "GLOBAL_MEAN = np.mean( train['is_duplicate'] ) \n",
        "print( 'Mean is_duplicated', GLOBAL_MEAN )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "20371f8c-10f0-fc7a-b678-1f46797e5cf7"
      },
      "outputs": [],
      "source": [
        "test[['test_id','prob']].to_csv(\"sub1.csv\",index=False,header=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7d750349-9779-7034-8bff-34965ebaa309"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}