{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport sklearn as sklearn #scikit machine learning library for preprocessing, data-split, modeling and evaluation\nimport matplotlib.pyplot as plt #data visiualization\n%matplotlib inline \n#displaying graphs in jupyter\nimport seaborn as sns #displaying distribution, correlation graphs\n\n!pip install altair\n!pip install datapane\n\nimport altair as alt #declarative visualization\nimport datapane as dp #Datapane is an open source framework which makes it easy to build and share reports using Python\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Quora Insincere Questions Problem\n\nThe steps followed in this problem:\n\n1. Loading the dataset\n2. Preprocessing of data\n* Lemmatizing\n* Tokenizing\n* Cleaning\n* Stopword\n* punctuations\n* Common words\n* URLs\n* HTML tags\n* Emojis"},{"metadata":{},"cell_type":"markdown","source":"# Loading the train, test and submission data"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import time\n%time\n\ntraining_data = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest_data = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\nsubmission_data=pd.read_csv('../input/quora-insincere-questions-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Check for missing values"},{"metadata":{"trusted":true},"cell_type":"code","source":"training_data.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are approx.1.3 million rows and 3 columns in the train dataset. No missing values. Data size 30MB.\n"},{"metadata":{},"cell_type":"markdown","source":"# First 5 records of train data"},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('max_colwidth',100)\ntraining_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = list(training_data.columns)\nprint(labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* qid - unique question identifier\n* question_text - Quora question text\n* target - \"insincere\" has a value of 1, otherwise is 0"},{"metadata":{},"cell_type":"markdown","source":"#  First 5 records of test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('max_colwidth',200)\ntest_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#  First 5 records of submission data"},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('max_colwidth',200)\nsubmission_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# First 5 Insincere Questions"},{"metadata":{"trusted":true},"cell_type":"code","source":"training_data[training_data.target==1][:5]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# First 5 Sincere Questions"},{"metadata":{"trusted":true},"cell_type":"code","source":"training_data[training_data.target==0][:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom seaborn import countplot\nfrom matplotlib.pyplot import suptitle\n%matplotlib inline\n\n#counts of sincere and insincere question texts:\ncount=training_data['target'].value_counts()\nprint('Total Counts of both sets'.format(),count)\n\nplt.figure(figsize=(10,5))\nsns.countplot(y=\"target\",palette =['maroon','blue'],data=training_data)\nplt.suptitle(\"Count values of Sincere Questions (0) & Insincere Questions (1)\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It's clearly unbalanced training data set"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn import preprocessing,metrics,manifold\nfrom sklearn.manifold import TSNE\nfrom sklearn.model_selection import train_test_split,cross_val_score,GridSearchCV,cross_val_predict\nfrom imblearn.over_sampling import ADASYN,SMOTE\nfrom imblearn.under_sampling import NearMiss\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.svm import SVC\nimport collections\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.metrics import classification_report,roc_auc_score,roc_curve,r2_score,recall_score,confusion_matrix,precision_recall_curve\nfrom collections import Counter\nfrom sklearn.model_selection import StratifiedKFold,KFold,StratifiedShuffleSplit\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nstop_words = stopwords.words('english')\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA, TruncatedSVD,SparsePCA\nfrom sklearn.metrics import classification_report,confusion_matrix\nfrom nltk.tokenize import word_tokenize\nfrom collections import defaultdict\nfrom collections import Counter\nfrom wordcloud import WordCloud,STOPWORDS\nimport nltk\nfrom nltk.corpus import stopwords\nimport string","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sincere_questions = training_data[training_data['target']== 0]['question_text']\nprint(\"Quora Sincere Question Content\",sincere_questions[:1])\nprint(\"______________________________________________________________________________________________________________________________________________\")\ninsincere_questions = training_data[training_data['target']== 1]['question_text']\nprint(\"Quora Insincere Question Content\",insincere_questions[:1])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploratory Data Analysis of Text Data - Part I"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Analyse the count of words in each segment - both positive and negative questions\n#Function for checking word length\ndef cal_len(data):\n    return len(data)\n\n\ncount_sincere = sincere_questions.str.split().apply(lambda z:cal_len(z))\ncount_insincere = insincere_questions.str.split().apply(lambda z:cal_len(z))\nprint(\"Sincere Questions Content:\" + str(count_sincere))\nprint(\"Insincere Questions Content:\" + str(count_insincere))\n\nfig,(ax1,ax2)= plt.subplots(1,2,figsize=(20,5))\nsns.distplot(count_insincere,ax=ax1,color='Blue')\nax1.set_title(\"counts of words in target=1 (Insincere)\")\nsns.distplot(count_sincere,ax=ax2,color='Red')\nax2.set_title(\"counts of words in target=0 (Sincere)\")\nfig.suptitle(\"Distribution of words in the training data\")\nplt.show()\n\n#Create generic plotter with Seaborn\ndef plot_count(count_ones,count_zeros,title_1,title_2,subtitle,figsize):\n    fig,(ax1,ax2)=plt.subplots(1,2,figsize=figsize)\n    sns.distplot(count_zeros,ax=ax1,color='Blue')\n    ax1.set_title(title_1)\n    sns.distplot(count_ones,ax=ax2,color='Red')\n    ax2.set_title(title_2)\n    fig.suptitle(subtitle)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Count Punctuations/Stopwords/Codes and other semantic datatypes\n#Using the \"generic_plotter\" function.\n\ncount_sincere_punctuations= sincere_questions.apply(lambda z: len([c for c in str(z) if c in string.punctuation]))\ncount_insincere_punctuations= insincere_questions.apply(lambda z:len([c for c in str(z) if c in string.punctuation]))\nplot_count(count_sincere_punctuations,count_insincere_punctuations,\n           \"Sincere Questions-Punctuations\",\"Insincere Questions-Punctuations\",\"Word Punctuation Analysis\",figsize=(15,5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Analyse Stopwords\nxlim=(0,1)\nylim=(0,450)\n\nstops=set(stopwords.words('english'))\ncount_sincere_stops= sincere_questions.apply(lambda z : np.mean([len(z) for w in str(z).split()]))\ncount_insincere_stops= insincere_questions.apply(lambda z : np.mean([len(z) for w in str(z).split()]))\nplot_count(count_sincere_stops,count_insincere_stops,\"Sincere Questions-Stopwords\",\"Insincere Questions-Stopwords\",\n           \"Question Text - Stopwords Analysis\",figsize=(20,5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#WordCloud Visualizations\n#Method for creating wordclouds\n\ndef display_cloud(data,color,figsize):\n    plt.subplots(figsize=figsize)\n    wc = WordCloud(stopwords=STOPWORDS,background_color=\"white\", contour_width=2, contour_color=color,\n                   max_words=2000, max_font_size=256,\n                   random_state=42)\n    wc.generate(' '.join(data))\n    plt.imshow(wc, interpolation=\"bilinear\")\n    plt.axis('off')\n    plt.show()\n    \ndisplay_cloud(training_data['question_text'],color='red',figsize=(15,15))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Wordlcouds for Insincere Questions Text\n\ndisplay_cloud(insincere_questions,'blue',figsize=(15,15))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Simplified counter function\ndef create_corpus(x=0):\n    corpus=[]\n    for x in training_data[training_data['target']==x]['question_text'].str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus\n\ncorpus=create_corpus(x=0)\ncounter=Counter(corpus)\nmost=counter.most_common()\nx=[]\ny=[]\nfor word,count in most[:100]:\n    if (word not in stops) :\n        x.append(word)\n        y.append(count)\n        \nplt.figure(figsize=(15,10))\nsns.barplot(x=y,y=x)\nplt.title(\"Most frequent words in descending order\")\nplt.xlabel(\"frequency\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Inference From EDA Part-I\n\nThe following can be inferred from the data:\n\n- The dataset is unbalanced\n- The dataset contains unequal number of semantics and punctuations for sincere and insincere questions text \n- The dataset contains redundant words \n- Stopwords are present in a equal distribution in the dataset\n\nWe have to do lots of cleaning!"},{"metadata":{},"cell_type":"markdown","source":"## Statistical Analysis-II\n\nIn this context , we will be exploring further into the analysis part. This would allow us to have a better idea which part of the data requires\nremoval and which part can be transformed before applying any model on it.\n\nGram analysis is an essential tool which forms the base of preparing a common bag of words model containing relevant data. This process implies that we are taking into consideration which words are present in conjunction with other words with a maximum frequency in the dataset. Grams can be n-ary implying that we can have many gram analysis taking n-words together.For example: a Ternary Gram Analysis(Tri-gram) includes analysing sentences which have 3 words occuring together at a higher frequency."},{"metadata":{"trusted":true},"cell_type":"code","source":"#Gram analysis on Training set- Bigram and Trigram\nstopword=set(stopwords.words('english'))\ndef gram_analysis(data,gram):\n    tokens=[t for t in data.lower().split(\" \") if t!=\"\" if t not in stopword]\n    ngrams=zip(*[tokens[i:] for i in range(gram)])\n    final_tokens=[\" \".join(z) for z in ngrams]\n    return final_tokens\n\n\n#Create frequency grams for analysis\ndef create_dict(data,grams):\n    freq_dict=defaultdict(int)\n    for sentence in data:\n        for tokens in gram_analysis(sentence,grams):\n            freq_dict[tokens]+=1\n    return freq_dict\n\n\ndef horizontal_bar_chart(df, color):\n    trace = go.Bar(\n        y=df[\"n_gram_words\"].values[::-1],\n        x=df[\"n_gram_frequency\"].values[::-1],\n        showlegend=False,\n        orientation = 'h',\n        marker=dict(\n            color=color,\n        ),\n    )\n    return trace\n\n\n\ndef create_new_df(freq_dict,):\n    freq_df=pd.DataFrame(sorted(freq_dict.items(),key=lambda z:z[1])[::-1])\n    freq_df.columns=['n_gram_words','n_gram_frequency']\n    #print(freq_df.head())\n    #plt.barh(freq_df['n_gram_words'][:20],freq_df['n_gram_frequency'][:20],linewidth=0.3)\n    #plt.show()\n    trace=horizontal_bar_chart(freq_df[:20],'orange')\n    return trace\n    \ndef plot_grams(trace_zero,trace_one):\n    fig = tools.make_subplots(rows=1, cols=2, vertical_spacing=0.04,\n                          subplot_titles=[\"Frequent words of positive reviews\", \n                                          \"Frequent words of negative reviews\"])\n    fig.append_trace(trace_zero, 1, 1)\n    fig.append_trace(trace_ones, 1, 2)\n    fig['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\n    py.iplot(fig, filename='word-plots')\n    \n    \ntrain_sin= sincere_questions\ntrain_insin= insincere_questions\n\nprint(\"Bi-gram analysis\")\nfreq_train_sin=create_dict(train_sin[:200],2)\n#print(freq_train_sin)\ntrace_zero=create_new_df(freq_train_sin)\n\nfreq_train_insin=create_dict(train_insin[:200],2)\n#print(freq_train_insin)\ntrace_ones=create_new_df(freq_train_insin)\n\nplot_grams(trace_zero,trace_ones)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Tri-gram analysis\")\nfreq_train_zero=create_dict(train_df_zero[:200],3)\n#print(freq_train_df_zero)\ntrace_zero=create_new_df(freq_train_df_zero)\nfreq_train_df_ones=create_dict(train_df_ones[:200],3)\n#print(freq_train_df_zero)\ntrace_ones=create_new_df(freq_train_df_ones)\nplot_grams(trace_zero,trace_ones)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}