{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\ntrain_df =pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ninsincere=train_df[['qid','question_text']].where(train_df['target']==1 & ~pd.isnull(train_df['question_text']))\ninsincere =insincere.dropna()\nprint (f\"Total questions {train_df.shape[0]}\")\nprint(f\"Total insincere questions {insincere.shape[0]}\")\ninsincere.head()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Lets clean the question_text now"},{"metadata":{"trusted":true},"cell_type":"code","source":"import nltk\nimport re\nfrom nltk.corpus import stopwords\nimport string\nfrom nltk.tokenize import sent_tokenize, word_tokenize   \nfrom nltk.tokenize import WordPunctTokenizer\nimport gensim\nfrom nltk.stem import WordNetLemmatizer, SnowballStemmer\n\npunc = WordPunctTokenizer()\nlemmatizer = WordNetLemmatizer() \nstop = stopwords.words('english')\nexclude = set(string.punctuation)\n\ndef clean_text(text):\n    word_tokens = (word_tokenize(text))\n    remove_stop=[w.lower() for w in word_tokens if w.lower() not in stop]\n    remove_punct=[c for c in remove_stop if c not in exclude and len(c)>3]\n    clean =\" \".join([re.sub(r'[^a-zA-Z0-9]','',i) for i in remove_punct ])\n    lemma=lemmatizer.lemmatize(clean)\n    return lemma\n\ninsincere['clean_question']=insincere['question_text'].map(clean_text)\ninsincere['clean_question']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets find out the total no of words , bigrams and trigrams\nword_freq={}\nfor w in insincere['clean_question'].values.tolist():\n    for word in w.split(\" \"):\n        if word  in word_freq.keys():\n            word_freq[word]+=1\n        else:\n            word_freq[word]=1\nprint (len(word_freq))\nlists = sorted(word_freq.items())\n\n\nwords_df=pd.DataFrame(lists,columns=['words','frequency']).sort_values(by='frequency',ascending=False)\nwords_df.set_index('words')[:50].plot(kind='bar',figsize=(20,10),title='Frequency Dist For Top 20 Words');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#lets find out the most frequent bigrams\nimport nltk\nfrom nltk.util import ngrams\nfrom nltk.collocations import BigramCollocationFinder\nw_list=insincere['clean_question'].map(lambda text:word_tokenize(text))\nbigm_freq_dict={}\nbigrams= (nltk.bigrams(w) for w in w_list)\n\n\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bigr_freq_list=[]\nbigr_freq_dict={}\ndef get_bigrams():\n    for bi in bigrams:\n        yield bi\nbig_obj=get_bigrams()   \nfor bigr in big_obj:\n    bigrams  = next(bigr)\n    if bigrams not in bigr_freq_list:\n        bigr_freq_dict[bigrams]=0\n    else:\n        bigr_freq_dict[bigrams]+=1\n#for i in w_list:\n #   word_fd = nltk.FreqDist(i)\n #   bigram_fd = nltk.FreqDist(nltk.bigrams(i))\n #   bigr_freq_list.extend(bigram_fd.most_common())\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}