{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "cf2cb403-001f-7678-c5fc-fb5b5ce30d05"
      },
      "source": [
        "Learning sklearn and trying to create word in common and tfidf features."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9f47b426-fc42-4a07-9593-5f9c7784ae27"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "import sklearn.feature_extraction.text\n",
        "import sklearn.pipeline\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d77a238b-f31b-accc-95f1-fcf8fd4b5320"
      },
      "outputs": [],
      "source": [
        "train = pd.read_csv('../input/train.csv').fillna('')\n",
        "test = pd.read_csv('../input/test.csv').fillna('')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b60b4087-a791-d525-36aa-0aee6e9c48e3"
      },
      "outputs": [],
      "source": [
        "questions = np.concatenate([\n",
        "    train.question1, train.question2, test.question1, test.question2,\n",
        "])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ca1f1a81-f38a-220f-3ef3-edfad2286ed5"
      },
      "outputs": [],
      "source": [
        "binary_count_vectorizer = sklearn.feature_extraction.text.CountVectorizer(binary=True).fit(questions)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "10588d40-5f24-e78b-e089-a412b8b16b75"
      },
      "outputs": [],
      "source": [
        "tfidf_transformer = sklearn.pipeline.make_pipeline(\n",
        "    sklearn.feature_extraction.text.CountVectorizer(binary=True),\n",
        "    sklearn.feature_extraction.text.TfidfTransformer()\n",
        ").fit(questions)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "12ad78f7-f3f0-65d1-39e8-a0ff1e5c237d"
      },
      "outputs": [],
      "source": [
        "def compute_overlap_rate(df, transformer):\n",
        "    q1_one_hot = transformer.transform(df['question1'])\n",
        "    q2_one_hot = transformer.transform(df['question2'])\n",
        "\n",
        "    q1_weights = q1_one_hot.sum(axis=1)\n",
        "    q2_weights = q2_one_hot.sum(axis=1)\n",
        "    q1_q2_overlap_weights = q1_one_hot.multiply(q2_one_hot).sum(axis=1)\n",
        "    overlap_rates = q1_q2_overlap_weights / (q1_weights + q2_weights)\n",
        "    return overlap_rates"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1064bee5-1afb-c088-6fb8-d26ffb810994"
      },
      "outputs": [],
      "source": [
        "def build_overlap_feature(df):\n",
        "    df['overlap_rates_one_hot'] = compute_overlap_rate(df, binary_count_vectorizer)\n",
        "    df['overlap_rates_tfidf'] = compute_overlap_rate(df, tfidf_transformer)\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f822d006-4173-470c-3237-eceac6c0219f"
      },
      "outputs": [],
      "source": [
        "train = build_overlap_feature(train)\n",
        "test = build_overlap_feature(test)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8406d202-bfa5-f7bf-9e15-a48cf9529b12"
      },
      "outputs": [],
      "source": [
        "test = test.fillna(0)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cb44de13-21ca-cdd4-c8b3-612b83f28e44"
      },
      "outputs": [],
      "source": [
        "train[['is_duplicate', 'overlap_rates_one_hot', 'overlap_rates_tfidf'\n",
        "       ]].to_csv('train_features.csv', index=False)\n",
        "test[['test_id', 'overlap_rates_one_hot', 'overlap_rates_tfidf']].to_csv(\n",
        "    'test_features.csv', index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}