{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\ndata = pd.read_csv('../input/train.csv')\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2dc7f7b42ab751411c7abd0c446417b386fcc765"},"cell_type":"code","source":"data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce3742edb8fdfed02451924e675861a84ec52a51"},"cell_type":"code","source":"data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"564232c7767e3ff93003e8cfd5781bf699e0b47b"},"cell_type":"code","source":"target0 = data[data['target'] == 0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"995c9084a4661403c2f751f7f0357e2297030eaf"},"cell_type":"code","source":"target1 = data[data['target']==1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0974f418af9c405321af4f2ba62bf9280ed462a"},"cell_type":"code","source":"from wordcloud import WordCloud","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93cffad3569d324cf3aff6c63e02d9885d67f1f2"},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb88e55c783cda72e93348f0f2b9fba91a262d86"},"cell_type":"code","source":"#create wordcloud for target 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56521d46eb0b8ddcfdae5dee5148ff5f2b8d2fb9"},"cell_type":"code","source":"doc0 = target0['question_text']\nwc0 = WordCloud(background_color='white').generate(''.join(doc0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"9d31eb048779d75b98755ba9f92af7aa48f1b85d"},"cell_type":"code","source":"plt.imshow(wc0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6473764721b8942379a19b6e5ca71d0997b99f4"},"cell_type":"code","source":"#wordcloud for target 1\ndoc1 = target1['question_text']\nwc1 = WordCloud(background_color='white').generate(''.join(doc1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"e9b8595b78ca111049b20278742aceed4b20c60e"},"cell_type":"code","source":"plt.imshow(wc1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f726cfa9914d3cd9e12f0e183835c567443550ab"},"cell_type":"code","source":"import nltk","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"304ac585ce33684333154d686a4d642e5116180e"},"cell_type":"code","source":"#data cleaning\nstopwords = nltk.corpus.stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a925675ac1a5c907da16099ab31cba780fef9255"},"cell_type":"code","source":"stemmer = nltk.stem.PorterStemmer()\n\ndef clean_sentence(doc):\n    words = doc.split(' ')\n    words_clean = [stemmer.stem(word) for word in words if word not in stopwords]\n    doc_clean = ' '.join(words_clean)\n    return doc_clean\n\ndocs = data['question_text'].str.lower().str.replace('[^a-z ]','')\ndocs_clean = docs.apply(clean_sentence)\n\ndocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bbd6c65fefe3f9f2f15148a262b2c4e06948ef7"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(docs_clean, data['target'], test_size=0.2, random_state=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2fb8604362526abc7eb6c1fbd12a5b7bb3ec283"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3e58587528d3b97e99208344897f12a9261232e"},"cell_type":"code","source":"vectorizer = CountVectorizer(min_df=50).fit(X_train)\nX_train = vectorizer.transform(X_train)\nX_test = vectorizer.transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33a14e7d006772923002f6eb48b46e62465b5f29"},"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB,MultinomialNB,BernoulliNB","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"297904ea450b2fd7f216e416d27b15d0f867587c"},"cell_type":"code","source":"from sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f5053cf6398af6a88da932e59bd269882d03a33"},"cell_type":"code","source":"model_mnb = MultinomialNB().fit(X_train,y_train)\ntest_pred = model_mnb.predict(X_test)\nprint(f1_score(y_test,test_pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2a640b6b014fdf6f2d1c7fcb4b62c945e025561"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9affb98490c2c36775a4d9b4ae1022ed5237a0bc"},"cell_type":"code","source":"#TF - IDF","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21da60754f34378a88150b546f5c0a2525fa4b32"},"cell_type":"code","source":"#from sklearn.feature_extraction.text import TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bab8cceeb0cebb49b0ae46a13b031a8e598ecfa"},"cell_type":"code","source":"#X_train, X_test, y_train, y_test = train_test_split(docs_clean, data['target'], test_size=0.2, random_state=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d71c3bebd764af79d51f7c9155786fdf364fc7b6"},"cell_type":"code","source":"#tfidf = TfidfVectorizer().fit(X_train) #it supresses the weights \n#X_train = tfidf.transform(X_train)\n#X_test = tfidf.transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10726f5388012ac5fe2e7cedb699f80f8533e54c"},"cell_type":"code","source":"#model_mnb = BernoulliNB().fit(X_train,y_train)\n#test_pred = model_mnb.predict(X_test)\n#print(f1_score(y_test,test_pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b15082482501107f17831220c7b8b21fa15965cb"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64c2e524a056924d20679d0dfd490ab569d7fd1b"},"cell_type":"code","source":"#now performing similar operation on Test data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b395d75408397cf21cadcb81d0bef5e9128180c"},"cell_type":"code","source":"dtest = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"645d1dec1e9beec4799534e088459004daf5534a"},"cell_type":"code","source":"dtest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3aaf2e6b7fe1d76f42d2301f4bc34cf79d0318e5"},"cell_type":"code","source":"def clean_sentence(doc):\n    words = doc.split(' ')\n    words_clean = [stemmer.stem(word) for word in words if word not in stopwords]\n    doc_clean_test = ' '.join(words_clean)\n    return doc_clean_test\n\ndocs = dtest['question_text'].str.lower().str.replace('[^a-z ]','')\ndocs_clean_test = docs.apply(clean_sentence)\n\ndocs_clean_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afa296dd7303539ed5aad64362d867de0daade5e"},"cell_type":"code","source":"test_data = vectorizer.transform(docs_clean_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39cdfffda86dd91261aad832febeb69bb613d315"},"cell_type":"code","source":"test_predict = model_mnb.predict(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"861672abe2d56b838ebcb13a0a7f6065cf0aa9e2"},"cell_type":"code","source":"#Final_pred = pd.DataFrame(test_predict)\n#Final_pred.to_csv('Final_pred.csv')\n\ndata_to_submit = pd.DataFrame({\n    'qid':dtest['qid'],\n    'prediction':test_predict\n})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2202ba742fcc4b70bb3d624117c85f4ee1dc82e8"},"cell_type":"code","source":"data_to_submit.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}