{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport nltk\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom wordcloud import WordCloud\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import f1_score,accuracy_score\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"data= pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ndata.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['target'].value_counts(normalize=True)*100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# For Understanding purpose\ndocs = data[\"question_text\"].str.lower()\n# Corpus is called as collection of docs\n# Document is collection of terms\n# Terms are collection of words\n# Corpus ---> Documents -->Terms --> Words\ndocs = docs.str.replace(\"[^a-z\\s#@]\", \"\")  #Retain alphabets, spaces, hastags, @ & spaces and remove everything else\ndocs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#nltk.download(\"stopwords\")\ndocs= data[data['target']==0][\"question_text\"].str.lower()\nwc= WordCloud().generate(\" \".join(docs))\nplt.figure(figsize=(14,4))\nplt.imshow(wc)\nplt.axis(False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"docs= data[data['target']==1]['question_text'].str.lower()\nwc= WordCloud().generate(\" \".join(docs))\nplt.figure(figsize=(14,4))\nplt.imshow(wc)\nplt.axis(False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"stopwords= nltk.corpus.stopwords.words(\"english\")\nstopwords.extend([\"\"]) #Extend custom stopwords\nstemmer= nltk.stem.PorterStemmer() # identifying root form of the word\nimport re\n\ndef clean_doc(doc):\n    doc= doc.lower()\n    doc= re.sub('[^a-z\\s]',\"\",doc)\n    words = doc.split(\" \")\n    words_imp= [stemmer.stem(word) for word in words if word not in stopwords]\n    doc_cleaned= \" \".join(words_imp)\n    return doc_cleaned","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer\n#df_dtm= pd.DataFrame(dtm.toarray(), columns=vectorizer.get_feature_names()) --> Memory error\n## Rows ---> Documents\n## Columns ---> tERMS\n## Values --> Frequency of each term in a document\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train,X_test,y_train,y_test= train_test_split(data[\"question_text\"].apply(clean_doc)\n                                                ,data[\"target\"],test_size=0.8,random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vectorizer = CountVectorizer(min_df=10).fit(X_train)\ndtm_train= vectorizer.transform(X_train)\ndtm_validate=vectorizer.transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model=MultinomialNB().fit(dtm_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred= model.predict(dtm_validate)\nfrom sklearn.metrics import f1_score,accuracy_score\nprint(accuracy_score(y_test,y_pred))\nprint(f1_score(y_test,y_pred))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}