{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gensim\nimport nltk\nimport os\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"path = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(path,binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"0c64450a16dcde38cc02d5184bceb15505f775d4"},"cell_type":"code","source":"list(embeddings['modi'][:5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a65c7940004a37c85c9362bbd1b5d4cc4c15ce38"},"cell_type":"code","source":"pd.Series(embeddings['modi'][:5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e5c9a6e7a6b81e67de6bd7891efcd7a1e716f84"},"cell_type":"code","source":"embeddings.most_similar('modi',topn=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfb87e0c07bd055d6aa56e8c975218c6c3fdd585"},"cell_type":"code","source":"url = 'https://bit.ly/2S2yXEd'\ndata = pd.read_csv(url)\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe8a38d1926b1b46193342e5a7b2c68e1d7a6fd1"},"cell_type":"code","source":"doc1 = data.iloc[0,0]\nprint(doc1)\nprint(nltk.word_tokenize(doc1.lower()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdd170f9f88733f71938e601005770512d6ea55b"},"cell_type":"code","source":"docs = data['review']\ndocs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f628e572e3bebf1bbb2b253e50608ae4b0f4fa8"},"cell_type":"code","source":"words = nltk.word_tokenize(doc1.lower())\ntemp = pd.DataFrame()\nfor word in words:\n    try:\n        print(word,embeddings[word][:5])\n        temp = temp.append(pd.Series(embeddings[word][:5]),ignore_index=True)\n    except:\n        print(word,'is not there')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"docs = docs.str.lower().str.replace('[^a-z ]','')\ndocs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8cf37b281fdec0122d54481519166c5e9f1a2b6f"},"cell_type":"code","source":"from nltk.stem import PorterStemmer\nstemmer = PorterStemmer()\nstopwords  = nltk.corpus.stopwords.words('english')\n\ndef clean_doc(doc):\n    words = doc.split(' ')\n    words_clean = [word for word in words if word not in stopwords]\n    doc_clean= ' '.join(words_clean)\n    return doc_clean\n\ndocs_clean = docs.apply(clean_doc)\ndocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"docs_clean.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"167d97b394ed8b1e835ede39528474e036ad42fc"},"cell_type":"code","source":"docs_vectors =  pd.DataFrame()\n\nfor doc in docs_clean:\n    words = nltk.word_tokenize(doc)\n    temp =  pd.DataFrame()\n    for word in words:\n        try:\n            word_vec = embeddings[word]\n            temp = temp.append(pd.Series(word_vec),ignore_index=True)\n        except:\n            pass\n    docs_vectors=docs_vectors.append(temp.mean(),ignore_index=True)   \ndocs_vectors.shape    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd908aed9721ab376d686f577dedb274a4980233"},"cell_type":"code","source":"docs_vectors.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"342e7a9f23aae57c81f9a407add065e493f5bceb"},"cell_type":"code","source":"pd.isnull(docs_vectors).sum(axis=1).sort_values(ascending=False).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30a51b1dd9e55d90c1857772b5b99b15a9a8aeaf"},"cell_type":"code","source":"X = docs_vectors.drop([64,590])\nY = data['sentiment'].drop([64,590])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e44956d0cb42114b1d3195ff5bd469fb5179600b"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nxtrain,xtest,ytrain,ytest = train_test_split(X,Y,test_size=.2,random_state=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff80db9354445c6de31132b5a85b484c0cbc4639"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier,AdaBoostClassifier\nfrom sklearn.metrics import accuracy_score\nmodel = RandomForestClassifier(n_estimators=800)\nmodel.fit(xtrain,ytrain)\ntest_pred =  model.predict(xtest)\naccuracy_score(ytest,test_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5284bd8acb3f75bc2644071fa996e27a2c8665a6"},"cell_type":"code","source":"model = AdaBoostClassifier(n_estimators=800)\nmodel.fit(xtrain,ytrain)\ntest_pred =  model.predict(xtest)\naccuracy_score(ytest,test_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"880607bfe675a6f2fe95c4f4e9e1e455fd5cedb0"},"cell_type":"markdown","source":"**HOTSTAR - GO SOLO review Analysis**"},{"metadata":{"trusted":true,"_uuid":"fdedbcf80e73c5ce3e3f7d1244e8f0ff007ef4df"},"cell_type":"code","source":"url = 'https://bit.ly/2W21FY7'\ndata = pd.read_csv(url)\ndata.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9335aea60c141c619af96ced100cd29896a674c"},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dbb565e467fa7052cabec415e19513a09012f404"},"cell_type":"code","source":"docs = data.loc[:,'Lower_Case_Reviews']\nprint(docs.shape)\ndocs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1742da564ab0520d94ac90801d387d169d1abc55"},"cell_type":"code","source":"Y = data['Sentiment_Manual']\nY.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e8a42ac63181172ea59091c8e39be9ba0217333"},"cell_type":"code","source":"Y.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5e1180d39a672bc95e054c8ded5919ad4e47841"},"cell_type":"code","source":"docs = docs.str.lower().str.replace('[^a-z ]','')\ndocs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from nltk.stem import PorterStemmer\nstemmer = PorterStemmer()\nstopwords  = nltk.corpus.stopwords.words('english')\n\ndef clean_doc(doc):\n    words = doc.split(' ')\n    words_clean = [stemmer.stem(word) for word in words if word not in stopwords]\n    doc_clean= ' '.join(words_clean)\n    return doc_clean\n\ndocs_clean = docs.apply(clean_doc)\ndocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"992724cad6ee57878d4f4bdfaff670530884768d"},"cell_type":"code","source":"X = docs_clean \nX.shape,Y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a0dca41b4c67df8066edf70fd14c80abbd99274"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nxtrain,xtest,ytrain,ytest = train_test_split(X,Y,test_size=.2,random_state=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0127fcf423c233ec25ca3d857d73cb457a7c2c13"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\ncv = CountVectorizer(min_df=5)\ncv.fit(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ed9c23d643c0e304e25eb1b1bfbcf835b9dcf5d"},"cell_type":"code","source":"XTRAIN = cv.transform(xtrain)\nXTEST = cv.transform(xtest)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"68376e78be9f5b356f9428e58b885b3e6e8343bf"},"cell_type":"code","source":"XTRAIN = XTRAIN.toarray()\nXTEST = XTEST.toarray()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9baf5ce3ff974837d38299cb320661d7f5c87d5b"},"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier as dtc\nfrom sklearn.metrics import accuracy_score\nmodel = dtc(max_depth=10)\nmodel.fit(XTRAIN,ytrain)\nyp= model.predict(XTEST)\naccuracy_score(ytest,yp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c547edb7344052f26e77c00180a433348e817796"},"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB as mnb\nm1=mnb()\nm1.fit(XTRAIN,ytrain)\nyp1=m1.predict(XTEST)\naccuracy_score(ytest,yp1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd81e1ba7613d495e9f0d05be5821036a58981bd"},"cell_type":"code","source":"from sklearn.naive_bayes import BernoulliNB as bnb\nm2=bnb()\nm2.fit(XTRAIN,ytrain)\nyp2=m2.predict(XTEST)\naccuracy_score(ytest,yp2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60dcd325da4845a2ab25df0723525d9cda9c77b3"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ntv = TfidfVectorizer(min_df=5)\ntv.fit(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ea07ff6c0b9f9c8b83402f3470f33efdb0ce965"},"cell_type":"code","source":"XTRAIN = tv.transform(xtrain)\nXTEST = tv.transform(xtest)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a136f20294fe85ef2f7ba5573b87b5276a0c1aae"},"cell_type":"code","source":"XTRAIN = XTRAIN.toarray()\nXTEST = XTEST.toarray()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5922f4594768d0594646feb2e7f0f2946beef3dc"},"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB as mnb\nmod=mnb()\nmod.fit(XTRAIN,ytrain)\nypred=mod.predict(XTEST)\naccuracy_score(ytest,ypred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"stopwords  = nltk.corpus.stopwords.words('english')\n\ndef clean_doc(doc):\n    words = doc.split(' ')\n    words_clean = [word for word in words if word not in stopwords]\n    doc_clean= ' '.join(words_clean)\n    return doc_clean\n\ndocs_clean = docs.apply(clean_doc)\ndocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"13e4ce68f195c702de59f86bbe044248c0044506"},"cell_type":"code","source":"docs_vectors =  pd.DataFrame()\n\nfor doc in docs_clean:\n    words = nltk.word_tokenize(doc)\n    temp =  pd.DataFrame()\n    for word in words:\n        try:\n            word_vec = embeddings[word]\n            temp = temp.append(pd.Series(word_vec),ignore_index=True)\n        except:\n            pass\n    docs_vectors=docs_vectors.append(temp.mean(),ignore_index=True)   \ndocs_vectors.shape    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b717a8ce0aefb482be25a66a756ba5d12e08f3a4"},"cell_type":"code","source":"df = pd.concat([docs_vectors,Y],axis=1)\ndf.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3dc3524b5828d609020aaf8efcc318524c5faa9c"},"cell_type":"code","source":"df[df.iloc[:,0].isnull()].shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05630dd7f522bddbacf52411e75f88dc57d9342a"},"cell_type":"code","source":"df = df.dropna(axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1d458c8383e19f82a22991ae1dc237a2cf7bb84"},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d307b03daf1f36248ef53a576fa81bb82e29c826"},"cell_type":"code","source":"X = df.drop(['Sentiment_Manual'],axis=1)\nY = df['Sentiment_Manual']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3ac4c2d93584b055d2cfb532a7418d96a8e3dca"},"cell_type":"code","source":"X.shape,Y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ce0060a601e74f11489a91fa23f3dfe9d48d69c"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nxtrain,xtest,ytrain,ytest = train_test_split(X,Y,test_size=.2,random_state=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"820f8c0c59c47e17d20a239d6d36da38fa7f01e3"},"cell_type":"code","source":"xtrain.shape,ytrain.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec012a090a0ff62fadbd79e34ff15ca9a363837d"},"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier as dtc\nfrom sklearn.metrics import accuracy_score\nmodel = dtc(max_depth=10)\nmodel.fit(xtrain,ytrain)\nyp= model.predict(xtest)\naccuracy_score(ytest,yp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d29cfc7da47789b38f53623e3dc549f3271459e0"},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0965f8871c97a880be6a3cde0e71292d3600d69"},"cell_type":"code","source":"data.Sentiment_Manual.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2640196a712fa6277cc1f70db428d07acb1f6e82"},"cell_type":"code","source":"docs_clean.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbbbece8660a7bf137ae2ae15a6a5c541851a072"},"cell_type":"code","source":"from nltk.sentiment import SentimentIntensityAnalyzer\nanalyser = SentimentIntensityAnalyzer()\n\ndef get_sentiment(sentence,analyser=analyser):\n    score = analyser.polarity_scores(sentence)['compound']\n    if score > 0:\n        return 1\n    else:\n        return 0","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}