{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "01c00aec-633e-330d-9510-1a301f44882a"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "76ffb4af-cf25-d2eb-45ef-e89e46de8f51"
      },
      "outputs": [],
      "source": [
        "train_df = pd.read_csv(\"../input/train.csv\")\n",
        "test_df = pd.read_csv(\"../input/test.csv\")\n",
        "\n",
        "print(train_df.shape)\n",
        "print(test_df.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8be4a7b2-ac88-ab74-6fff-ddd0cd28700b"
      },
      "outputs": [],
      "source": [
        "train_df.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "22c116c2-73b3-e352-188a-1b5d68aa98f5"
      },
      "outputs": [],
      "source": [
        "test_df.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3c090c48-53c1-0d66-8b1c-5cf60a53ef19"
      },
      "outputs": [],
      "source": [
        "#print(train_df[train_df['question1']==\"\"])\n",
        "train_df[train_df['question2'].isnull()]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6fc24480-1cb3-b0b0-6d90-98dbfa8e7e32"
      },
      "outputs": [],
      "source": [
        "train_df[train_df['question1'].isnull()]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "15147170-51b0-b360-b359-2f9e71953f52"
      },
      "outputs": [],
      "source": [
        "train_df.dropna(inplace=True)\n",
        "print(len(train_df))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "44cda957-7568-cb57-daee-2ed6125ee83b"
      },
      "outputs": [],
      "source": [
        "from nltk.corpus import stopwords\n",
        "import string\n",
        "from nltk.stem.wordnet import WordNetLemmatizer\n",
        "from tqdm import tqdm\n",
        "from collections import OrderedDict as OD\n",
        "import nltk\n",
        "\n",
        "stop_words= list(stopwords.words('english'))\n",
        "punc_list= list(string.punctuation)\n",
        "lemma = WordNetLemmatizer()\n",
        "\n",
        "def clean_docs(question):\n",
        "    stop_words_cleaned= ' '.join([i for i in question.lower().split() if i not in stop_words])\n",
        "    punc_cleaned=''.join([i for i in stop_words_cleaned if i not in punc_list])\n",
        "    normalized= ' '.join([lemma.lemmatize(i) for i in punc_cleaned.split()])\n",
        "    #unique_words= ' '.join(OD.fromkeys(normalized.split()))\n",
        "    return normalized\n",
        "    \n",
        "\"\"\"\n",
        "def clean_sentence(sentence):\n",
        "    \"POS tag the sentence and then lemmatize the words.\"\n",
        "    global lemmatizer\n",
        "    words = sentence.lower().split()\n",
        "    pos_tagged = nltk.pos_tag(words)\n",
        "    pos_tagged_stems = [(lemmatizer.lemmatize(i[0]), i[1]) for i in pos_tagged]\n",
        "    return pos_tagged_stems\n",
        "\"\"\"\n",
        "\"\"\"\n",
        "for i in tqdm(range(len(train_df['question1'])),mininterval=15):\n",
        "    train_df.loc[train_df['question1']!=0,'q1_cleaned']=clean_docs(train_df['question1'][i])\n",
        "\"\"\"\n",
        "tqdm.pandas(mininterval=15, ncols=80, desc='Question1')\n",
        "train_df['q1_clean'] = train_df.question1.progress_apply(clean_docs)\n",
        "\n",
        "tqdm.pandas(mininterval=15, ncols=80, desc='Question2')\n",
        "train_df['q2_clean'] = train_df.question2.progress_apply(clean_docs)\n",
        "    \n",
        "    "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f7b5d304-3e00-4a16-8241-f4143fd520c0"
      },
      "outputs": [],
      "source": [
        "train_df.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7e288c2f-c4ac-c2d1-bb50-7319c0cadc32"
      },
      "outputs": [],
      "source": [
        "train_df[train_df['is_duplicate']==1].head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "090a0e39-7f17-cce7-fe0d-d88963dbbea1"
      },
      "outputs": [],
      "source": [
        "train_df[train_df['is_duplicate']==0].head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "676c87be-fa78-0496-55b8-965abea6e403"
      },
      "outputs": [],
      "source": [
        "train_df.isnull().any()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5da99011-822f-9b7e-6776-ce531e8bd749"
      },
      "outputs": [],
      "source": [
        "from sklearn.feature_extraction.text import TfidfVectorizer\n",
        "from sklearn.metrics.pairwise import cosine_similarity\n",
        "\n",
        "tfidf= TfidfVectorizer(min_df=1)\n",
        "\n",
        "similarity_list=[]\n",
        "\n",
        "train_df['combined_q'] = list(zip(train_df['q1_clean'],train_df['q2_clean']))\n",
        "\n",
        "def tf_idf(document):\n",
        "    try:\n",
        "        tfidf_matrix=tfidf.fit_transform(document)\n",
        "        return tfidf_matrix\n",
        "    except ValueError():\n",
        "        return 0\n",
        "tqdm.pandas(mininterval=15, ncols=80, desc='combined_q')\n",
        "train_df['tf_idf']= train_df['combined_q'].progress_apply(tf_idf)\n",
        "\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ffa4e909-6df3-5685-f382-3b957d09a4dc"
      },
      "outputs": [],
      "source": [
        "def get_cosine_sim(vectors):\n",
        "    return float(cosine_similarity(vectors[0],vectors[1]).A)\n",
        "\n",
        "train_df['tf_idf'].head()\n",
        "    "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7314e166-9a3a-902b-0703-2b76558e1006"
      },
      "outputs": [],
      "source": [
        "from sklearn.feature_extraction.text import TfidfVectorizer\n",
        "tfidf_vectorizer = TfidfVectorizer(min_df=1)\n",
        "\n",
        "train_df['combined_q']= list(zip(train_df['q1_clean'],train_df['q2_clean']))\n",
        "\n",
        "#print(train_df['combined_q'].head())\n",
        "\n",
        "def tfidf(docs):\n",
        "    try:\n",
        "        tfidf_matrix = tfidf_vectorizer.fit_transform(docs)\n",
        "        return tfidf_matrix\n",
        "    except ValueError:\n",
        "        return 0\n",
        "tqdm.pandas(mininterval=15, ncols=80, desc='combined_q')\n",
        "train_df['tf_idf'] = train_df.combined_q.progress_apply(tfidf)\n",
        "\n",
        "   "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3e040ace-9d37-a783-00c1-a1b88422e776"
      },
      "outputs": [],
      "source": [
        "from sklearn.metrics.pairwise import cosine_similarity\n",
        "#df_train['tf_idf'].drop()\n",
        "def cosine_sim(vectors):\n",
        "    try:\n",
        "        return float(cosine_similarity(vectors[0],vectors[1]))\n",
        "    except ValueError:\n",
        "        return 0\n",
        "tqdm.pandas(mininterval=15, ncols=80, desc='tf_idf')\n",
        "train_df['cosine_sim'] = train_df.tf_idf.progress_apply(cosine_sim)\n",
        "#print(cosine_sim(train_df['tf_idf'][1]))   \n",
        "#print()\n",
        "#print(cosine_similarity(train_df['tf_idf'][1][0],train_df['tf_idf'][1][1]))"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}