{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport io\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn import preprocessing,metrics,manifold\nfrom sklearn.manifold import TSNE\nfrom sklearn.model_selection import train_test_split,cross_val_score,GridSearchCV,cross_val_predict\nfrom imblearn.over_sampling import ADASYN,SMOTE\nfrom imblearn.under_sampling import NearMiss\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.svm import SVC\nimport collections\nimport matplotlib.patches as mpatches\nfrom sklearn.metrics import accuracy_score\n%matplotlib inline\nfrom sklearn.preprocessing import RobustScaler\nimport xgboost\nfrom imblearn.metrics import classification_report_imbalanced\nfrom sklearn.metrics import classification_report,roc_auc_score,roc_curve,r2_score,recall_score,confusion_matrix,precision_recall_curve\nfrom collections import Counter\nfrom sklearn.model_selection import StratifiedKFold,KFold,StratifiedShuffleSplit\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nstop_words = stopwords.words('english')\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA, TruncatedSVD,SparsePCA\nfrom sklearn.metrics import classification_report,confusion_matrix\nfrom nltk.tokenize import word_tokenize\nfrom collections import defaultdict\nfrom collections import Counter\nimport seaborn as sns\nfrom wordcloud import WordCloud,STOPWORDS\nimport nltk\nfrom nltk.corpus import stopwords\nimport string\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.target.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Sincere= train_df[train_df['target']== 0]['question_text']\nInsincere=train_df[train_df['target']== 1]['question_text']\nprint(\"First 5 samples of Sincere Questions\\n\".format(),Sincere[:5])\nprint(\"First 5 samples of Insincere Questions\\n\".format(),Insincere[:5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Count of Insincere and Sincere Questions\ntrain_df.target.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Analyse the count of words in each segment- both Sincere and Insincere\n#Function for checking word length\ndef cal_len(data):\n    return len(data)\n\n#Create generic plotter with Seaborn\ndef plot_count(count_ones,count_zeros,title_1,title_2,subtitle):\n    fig,(ax1,ax2)=plt.subplots(1,2,figsize=(15,5))\n    sns.distplot(count_zeros,ax=ax1,color='Blue')\n    ax1.set_title(title_1)\n    sns.distplot(count_ones,ax=ax2,color='Red')\n    ax2.set_title(title_2)\n    fig.suptitle(subtitle)\n    plt.show()    \n\n\ncount_good=train_df[train_df['target']== 0]['question_text']\ncount_bad=train_df[train_df['target']== 1]['question_text']\n\ncount_good_words=count_good.str.split().apply(lambda z:cal_len(z))\ncount_bad_words=count_bad.str.split().apply(lambda z:cal_len(z))\nprint(\"Sincere Question Words:\" + str(count_good_words))\nprint(\"Insincere Question Words:\" + str(count_bad_words))\nplot_count(count_good_words,count_bad_words,\"Sincere Question\",\"Insincere Question\",\n           \"Reviews Word Analysis\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Count Punctuations/Stopwords/Codes and other semantic datatypes\n#We will be using the \"generic_plotter\" function.\n\ncount_good_punctuations=count_good.apply(lambda z: len([c for c in str(z) if c in string.punctuation]))\ncount_bad_punctuations=count_bad.apply(lambda z:len([c for c in str(z) if c in string.punctuation]))\nplot_count(count_good_punctuations,count_bad_punctuations,\"Positive Review Punctuations\",\n           \"Negative Review Punctuations\",\"Reviews Word Punctuation Analysis\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Analyse Stopwords\n\ndef plot_count_1(count_ones,count_zeros,title_1,title_2,subtitle):\n    fig,(ax1,ax2)=plt.subplots(1,2,figsize=(15,5))\n    sns.distplot(count_zeros,ax=ax1,color='Blue')\n    ax1.set_title(title_1)\n    sns.distplot(count_ones,ax=ax2,color='Orange')\n    ax2.set_title(title_2)\n    fig.suptitle(subtitle)\n    plt.show()    \n\n\nstops=set(stopwords.words('english'))\ncount_good_stops=count_good.apply(lambda z : len([w for w in str(z).split() if w in \n                                 stops]))\ncount_bad_stops=count_bad.apply(lambda z : len([w for w in str(z).split() if w in \n                                 stops]))\nplot_count_1(count_good_stops,count_bad_stops,\"Positive Reviews Stopwords\",\n             \"Negative Reviews Stopwords\",\"Reviews Stopwords Analysis\")\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count_good[:2].apply(lambda z : [w for w in str(z).split() if w in \n                                 stopwords.words('english')])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\n## Checking number of Urls\ncount_good_urls=count_good.apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\ncount_bad_urls=count_bad.apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\n\nplot_count_1(count_good_stops,count_bad_stops,\"Positive Reviews URLs\",\n             \"Negative Reviews URLs\",\"Reviews URLs Analysis\")\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Simplified counter function\ndef create_corpus(data):\n    corpus=[]\n    for x in data.str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus\n\ncorpus=create_corpus(count_good)\ncorpus\ncounter=Counter(corpus)\nmost=counter.most_common()\nwords=[]\ncounts=[]\nfor word,count in most[:80]:\n    if (word not in stops) :\n        words.append(word)\n        counts.append(count)\nsns.barplot(x=counts,y=words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Simplified counter function\ndef create_corpus(data):\n    corpus=[]\n    for x in data.str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus\n\ncorpus=create_corpus(count_bad)\ncorpus\ncounter=Counter(corpus)\nmost=counter.most_common()\nwords=[]\ncounts=[]\nfor word,count in most[:80]:\n    if (word not in stops) :\n        words.append(word)\n        counts.append(count)\nsns.barplot(x=counts,y=words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# GRAM Statistics\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}