{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4b29202d-9d1b-7272-13e6-c93d408ce5fe"
      },
      "outputs": [],
      "source": [
        "#0.33719 on Public Leaderboard\n",
        "\n",
        "import pandas as pd\n",
        "import numpy as np\n",
        "import nltk\n",
        "from collections import Counter\n",
        "from nltk.corpus import stopwords\n",
        "from sklearn.metrics import log_loss\n",
        "from scipy.optimize import minimize\n",
        "stops = set(stopwords.words(\"english\"))\n",
        "import xgboost as xgb\n",
        "from sklearn.cross_validation import train_test_split\n",
        "import multiprocessing\n",
        "import difflib\n",
        "\n",
        "train = pd.read_csv('../input/train.csv')[:10000] #remove limit\n",
        "test = pd.read_csv('../input/test.csv')[:10000] #remove limit\n",
        "\n",
        "def diff_ratios(st1, st2):\n",
        "    seq = difflib.SequenceMatcher()\n",
        "    seq.set_seqs(str(st1).lower(), str(st2).lower())\n",
        "    return seq.ratio()\n",
        "\n",
        "def word_match_share(row):\n",
        "    q1words = {}\n",
        "    q2words = {}\n",
        "    for word in str(row['question1']).lower().split():\n",
        "        if word not in stops:\n",
        "            q1words[word] = 1\n",
        "    for word in str(row['question2']).lower().split():\n",
        "        if word not in stops:\n",
        "            q2words[word] = 1\n",
        "    if len(q1words) == 0 or len(q2words) == 0:\n",
        "        return 0\n",
        "    shared_words_in_q1 = [w for w in q1words.keys() if w in q2words]\n",
        "    shared_words_in_q2 = [w for w in q2words.keys() if w in q1words]\n",
        "    R = (len(shared_words_in_q1) + len(shared_words_in_q2))/(len(q1words) + len(q2words))\n",
        "    return R\n",
        "\n",
        "def get_weight(count, eps=500, min_count=2):\n",
        "    if count < min_count:\n",
        "        return 0\n",
        "    else:\n",
        "        return 1 / (count + eps)\n",
        "\n",
        "train_qs = pd.Series(train['question1'].tolist() + train['question2'].tolist()).astype(str)\n",
        "test_qs = pd.Series(test['question1'].tolist() + test['question2'].tolist()).astype(str)\n",
        "words = (\" \".join(train_qs)).lower().split() + (\" \".join(test_qs)).lower().split()\n",
        "counts = Counter(words)\n",
        "weights = {word: get_weight(count) for word, count in counts.items()}\n",
        "\n",
        "def tfidf_word_match_share(row):\n",
        "    q1words = {}\n",
        "    q2words = {}\n",
        "    for word in str(row['question1']).lower().split():\n",
        "        if word not in stops:\n",
        "            q1words[word] = 1\n",
        "    for word in str(row['question2']).lower().split():\n",
        "        if word not in stops:\n",
        "            q2words[word] = 1\n",
        "    if len(q1words) == 0 or len(q2words) == 0:\n",
        "        return 0\n",
        "    shared_weights = [weights.get(w, 0) for w in q1words.keys() if w in q2words] + [weights.get(w, 0) for w in q2words.keys() if w in q1words]\n",
        "    total_weights = [weights.get(w, 0) for w in q1words] + [weights.get(w, 0) for w in q2words]\n",
        "    \n",
        "    R = np.sum(shared_weights) / np.sum(total_weights)\n",
        "    return R\n",
        "\n",
        "def get_unigrams(que):\n",
        "    return [word for word in nltk.word_tokenize(que.lower()) if word not in stops]\n",
        "\n",
        "def get_common_unigrams(row):\n",
        "    return len( set(row[\"unigrams_ques1\"]).intersection(set(row[\"unigrams_ques2\"])) )\n",
        "\n",
        "def get_common_unigram_ratio(row):\n",
        "    return float(row[\"zunigrams_common_count\"]) / max(len( set(row[\"unigrams_ques1\"]).union(set(row[\"unigrams_ques2\"])) ),1)\n",
        "\n",
        "def get_bigrams(que):\n",
        "    return [i for i in nltk.ngrams(que, 2)]\n",
        "\n",
        "def get_common_bigrams(row):\n",
        "    return len( set(row[\"bigrams_ques1\"]).intersection(set(row[\"bigrams_ques2\"])) )\n",
        "\n",
        "def get_common_bigram_ratio(row):\n",
        "    return float(row[\"zbigrams_common_count\"]) / max(len( set(row[\"bigrams_ques1\"]).union(set(row[\"bigrams_ques2\"])) ),1)\n",
        "\n",
        "train['question1_nouns'] = train.question1.map(lambda x: [w for w, t in nltk.pos_tag(nltk.word_tokenize(str(x).lower())) if t[:1] in ['N']])\n",
        "train['question2_nouns'] = train.question2.map(lambda x: [w for w, t in nltk.pos_tag(nltk.word_tokenize(str(x).lower())) if t[:1] in ['N']])\n",
        "\n",
        "train['z_len1'] = train.question1.map(lambda x: len(str(x)))\n",
        "train['z_len2'] = train.question2.map(lambda x: len(str(x)))\n",
        "train['z_word_len1'] = train.question1.map(lambda x: len(str(x).split()))\n",
        "train['z_word_len2'] = train.question2.map(lambda x: len(str(x).split()))\n",
        "train['z_noun_match'] = train.apply(lambda r: sum([1 for w in r.question1_nouns if w in r.question2_nouns]), axis=1)\n",
        "\n",
        "train['z_match_ratio'] = train.apply(lambda r: diff_ratios(r.question1, r.question2), axis=1)\n",
        "\n",
        "train['z_word_match'] = train.apply(word_match_share, axis=1, raw=True)\n",
        "train['z_tfidf_word_match'] = train.apply(tfidf_word_match_share, axis=1, raw=True)\n",
        "\n",
        "train[\"unigrams_ques1\"] = train['question1'].apply(lambda x: get_unigrams(str(x)))\n",
        "train[\"unigrams_ques2\"] = train['question2'].apply(lambda x: get_unigrams(str(x)))\n",
        "train[\"zunigrams_common_count\"] = train.apply(lambda r: get_common_unigrams(r),axis=1)\n",
        "train[\"zunigrams_common_ratio\"] = train.apply(lambda r: get_common_unigram_ratio(r), axis=1)\n",
        "train[\"bigrams_ques1\"] = train[\"unigrams_ques1\"].apply(lambda x: get_bigrams(x))\n",
        "train[\"bigrams_ques2\"] = train[\"unigrams_ques2\"].apply(lambda x: get_bigrams(x)) \n",
        "train[\"zbigrams_common_count\"] = train.apply(lambda r: get_common_bigrams(r),axis=1)\n",
        "train[\"zbigrams_common_ratio\"] = train.apply(lambda r: get_common_bigram_ratio(r), axis=1)\n",
        "#-------------------------------------------------------------------------------------------------\n",
        "\n",
        "test['question1_nouns'] = test.question1.map(lambda x: [w for w, t in nltk.pos_tag(nltk.word_tokenize(str(x).lower())) if t[:1] in ['N']])\n",
        "test['question2_nouns'] = test.question2.map(lambda x: [w for w, t in nltk.pos_tag(nltk.word_tokenize(str(x).lower())) if t[:1] in ['N']])\n",
        "\n",
        "test['z_len1'] = test.question1.map(lambda x: len(str(x)))\n",
        "test['z_len2'] = test.question2.map(lambda x: len(str(x)))\n",
        "test['z_word_len1'] = test.question1.map(lambda x: len(str(x).split()))\n",
        "test['z_word_len2'] = test.question2.map(lambda x: len(str(x).split()))\n",
        "test['z_noun_match'] = test.apply(lambda r: sum([1 for w in r.question1_nouns if w in r.question2_nouns]), axis=1)\n",
        "\n",
        "test['z_match_ratio'] = test.apply(lambda r: diff_ratios(r.question1, r.question2), axis=1)\n",
        "\n",
        "test['z_word_match'] = test.apply(word_match_share, axis=1, raw=True)\n",
        "test['z_tfidf_word_match'] = test.apply(tfidf_word_match_share, axis=1, raw=True)\n",
        "\n",
        "test[\"unigrams_ques1\"] = test['question1'].apply(lambda x: get_unigrams(str(x)))\n",
        "test[\"unigrams_ques2\"] = test['question2'].apply(lambda x: get_unigrams(str(x)))\n",
        "test[\"zunigrams_common_count\"] = test.apply(lambda r: get_common_unigrams(r),axis=1)\n",
        "test[\"zunigrams_common_ratio\"] = test.apply(lambda r: get_common_unigram_ratio(r), axis=1)\n",
        "test[\"bigrams_ques1\"] = test[\"unigrams_ques1\"].apply(lambda x: get_bigrams(x))\n",
        "test[\"bigrams_ques2\"] = test[\"unigrams_ques2\"].apply(lambda x: get_bigrams(x)) \n",
        "test[\"zbigrams_common_count\"] = test.apply(lambda r: get_common_bigrams(r),axis=1)\n",
        "test[\"zbigrams_common_ratio\"] = test.apply(lambda r: get_common_bigram_ratio(r), axis=1)\n",
        "\n",
        "train = train.fillna(-1)\n",
        "test = test.fillna(-1)\n",
        "\n",
        "#train.to_csv('train.csv', index=False)\n",
        "#test.to_csv('test.csv', index=False)\n",
        "\n",
        "col = [c for c in train.columns if c[:1]=='z']\n",
        "\n",
        "pos_train = train[train['is_duplicate'] == 1]\n",
        "neg_train = train[train['is_duplicate'] == 0]\n",
        "p = 0.165\n",
        "scale = ((len(pos_train) / (len(pos_train) + len(neg_train))) / p) - 1\n",
        "while scale > 1:\n",
        "    neg_train = pd.concat([neg_train, neg_train])\n",
        "    scale -=1\n",
        "neg_train = pd.concat([neg_train, neg_train[:int(scale * len(neg_train))]])\n",
        "train = pd.concat([pos_train, neg_train])\n",
        "\n",
        "x_train, x_valid, y_train, y_valid = train_test_split(train[col], train['is_duplicate'], test_size=0.2, random_state=0)\n",
        "\n",
        "params = {}\n",
        "params[\"objective\"] = \"binary:logistic\"\n",
        "params['eval_metric'] = 'logloss'\n",
        "params[\"eta\"] = 0.02\n",
        "params[\"subsample\"] = 0.7\n",
        "params[\"min_child_weight\"] = 1\n",
        "params[\"colsample_bytree\"] = 0.7\n",
        "params[\"max_depth\"] = 4\n",
        "params[\"silent\"] = 1\n",
        "params[\"seed\"] = 1632\n",
        "\n",
        "d_train = xgb.DMatrix(x_train, label=y_train)\n",
        "d_valid = xgb.DMatrix(x_valid, label=y_valid)\n",
        "watchlist = [(d_train, 'train'), (d_valid, 'valid')]\n",
        "bst = xgb.train(params, d_train, 500, watchlist, early_stopping_rounds=50, verbose_eval=100) #change to 5000\n",
        "\n",
        "d_test = xgb.DMatrix(test[col])\n",
        "p_test = bst.predict(d_test)\n",
        "\n",
        "sub = pd.DataFrame()\n",
        "sub['test_id'] = test['test_id']\n",
        "sub['is_duplicate'] = p_test\n",
        "\n",
        "#df['is_duplicate'] = df['is_duplicate'].map(lambda x: 0.000000000001 if x < 0.0001 else x)\n",
        "#df['is_duplicate'] = df['is_duplicate'].map(lambda x: 0.999999999999 if x > 0.98 else x)\n",
        "\n",
        "sub.to_csv('z05_submission_xgb_03.csv', index=False)\n",
        "print(log_loss(train.is_duplicate, bst.predict(xgb.DMatrix(train[col]))))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ca1e2fee-1dd8-80ef-e3a3-ac995b257fa7"
      },
      "outputs": [],
      "source": [
        "import matplotlib.pyplot as plt\n",
        "import seaborn as sns\n",
        "%matplotlib inline\n",
        "\n",
        "plt.rcParams['figure.figsize'] = (7.0, 7.0)\n",
        "xgb.plot_importance(bst); plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a74702ae-a6ff-6873-03e0-77f1464ad967"
      },
      "outputs": [],
      "source": [
        "plt.rcParams['figure.figsize'] = (20.0, 20.0)\n",
        "xgb.plot_tree(bst, num_trees=0); plt.show()"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}