{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gensim\nimport nltk","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"print(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))\npath = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(path , binary = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"588ef679eb85503dca4ce41a4cbe38ba1aad8e1a"},"cell_type":"code","source":"url = 'https://raw.githubusercontent.com/skathirmani/datasets/master/hotstar.allreviews_Sentiments.csv'\ndf_hotstar = pd.read_csv(url)\ndf_hotstar['Sentiment_Manual'].head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76bc018530dad9649592c77c0580477435fbf929"},"cell_type":"code","source":"df_hotstar.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc2e440ee06aaabc7b5b038f7181d1689aa023aa"},"cell_type":"code","source":"df_hotstar['Sentiment_Manual'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c9024f9e75d8721563d9e2d161bc36b7b8901be"},"cell_type":"code","source":"nltk.download('stopwords')\nnltk.download('vader_lexicon')\nnltk.download('punkt')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"943cec5341e364a043f7f0b554e93a83b75fbc69"},"cell_type":"markdown","source":"## Word Cloud"},{"metadata":{"trusted":true,"_uuid":"4529ee32d817864e8f0b1ea5f2434201255c271b"},"cell_type":"code","source":"Neutral = df_hotstar[df_hotstar['Sentiment_Manual'] == 'Neutral']\nPositive = df_hotstar[df_hotstar['Sentiment_Manual'] == 'Positive']\nNegative = df_hotstar[df_hotstar['Sentiment_Manual'] == 'Negative']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba106812c1ad11aab4e649cefd223c0dd5fa6315"},"cell_type":"code","source":"Docs1 = Neutral['Lower_Case_Reviews']\nprint(len(Docs1))\n\nDocs2 = Positive['Lower_Case_Reviews']\nprint(len(Docs2))\n\nDocs3 = Negative['Lower_Case_Reviews']\nprint(len(Docs3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7104dbeedb7c52902f15adad8080c7eb1ee20ff"},"cell_type":"code","source":"! pip install wordcloud","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec1a27d6faf1cb0f1d9bc3443dc23a2ff4d654e4"},"cell_type":"code","source":"from wordcloud import WordCloud\nimport matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfc9d34dc69bcc8f1d9785a03bf8ec20ab01191a"},"cell_type":"code","source":"StopWords = nltk.corpus.stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2e334e1afbdfd51df4293fa20847067d45eb2cc"},"cell_type":"code","source":"WC_Neutral = WordCloud(background_color = 'white' , stopwords = StopWords).generate('' . join(Docs1))\nplt.imshow(WC_Neutral)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07da708027d5f22908dc4d23d26b00546584b2ef"},"cell_type":"code","source":"WC_Positive = WordCloud(background_color = 'white' , stopwords = StopWords).generate('' . join(Docs2))\nplt.imshow(WC_Positive)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55f0533d962bda3b246b1d945ee8691bde840f11"},"cell_type":"code","source":"WC_Negative = WordCloud(background_color = 'white' , stopwords = StopWords).generate('' . join(Docs3))\nplt.imshow(WC_Negative)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fd438ee957222b87df2f5d842b8f4bdbce3d5196"},"cell_type":"markdown","source":"## Data Cleaning"},{"metadata":{"trusted":true,"_uuid":"288430aeb63dd63cf89f50cc54d468e1ecdb9ce4"},"cell_type":"code","source":"Docs = df_hotstar['Lower_Case_Reviews']\nDocs = Docs.str.replace('-' , ' ').str.lower().str.replace('[^a-z ]' , ' ')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24be7842ea77b2e442adf9478f4059d6066bbaa2"},"cell_type":"code","source":"Docs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f57319fcf397aa227f7f2b2da7073d1f4a2bcb27"},"cell_type":"code","source":"StopWords = nltk.corpus.stopwords.words('english')\nclean_sentence = lambda doc: ' '.join([word for word in nltk.word_tokenize(doc) if word not in StopWords])\nDocs_clean = Docs.apply(clean_sentence)\nDocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cae7dc25d4ef87921ad9f2caec4d1843ecad9d38"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import MultinomialNB , BernoulliNB\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a419f093539b98bb4f3be53c4d848b13d6bbdfb0"},"cell_type":"markdown","source":"## Train-test split"},{"metadata":{"trusted":true,"_uuid":"288d31424d287a1a939974bcb5ccf77a0c9c2c16"},"cell_type":"code","source":"x_train , x_test , y_train , y_test = train_test_split(Docs_clean , df_hotstar['Sentiment_Manual'] , \n                                                       test_size = 0.2 , random_state = 100)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"011bc436879a091ea53343c2ec011f7c6edc63bb"},"cell_type":"markdown","source":"## Count Vectorizer"},{"metadata":{"trusted":true,"_uuid":"0205ab7ab2942004f5206e9de7dcfecb1cd2949e"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"539e5832625c1daff77f28395f0bca4de39ec42a"},"cell_type":"code","source":"vec = CountVectorizer(min_df = 5).fit(x_train)\nx_train = vec.transform(x_train)\nx_test = vec.transform(x_test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"523318860956996592f0582f44e82d4679d38c9b"},"cell_type":"markdown","source":"## Multinomial Naive Bayes Classification"},{"metadata":{"trusted":true,"_uuid":"96109c23dd4ceb29c22b99bd7fe13ebcc48c7b98"},"cell_type":"code","source":"model_mnb = MultinomialNB().fit(x_train , y_train)\ntest_pred_mnb = model_mnb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_mnb))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"916b6ce7220d8c75b701f1f7746ee171ecd7a8e3"},"cell_type":"markdown","source":"## Ada Boost Count Vectorizer"},{"metadata":{"trusted":true,"_uuid":"d6f85e2164ac34db0f2ca54986cd850ad069bc5c"},"cell_type":"code","source":"model_ab = AdaBoostClassifier(n_estimators = 100 , random_state = 99).fit(x_train , y_train)\ntest_pred_ab = model_ab.predict(x_test)\nprint(accuracy_score(y_test , test_pred_ab))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7c1fa2c451bb238f09820a2a18458eaf65688d5"},"cell_type":"markdown","source":"## Random Forest Count Vectorizer"},{"metadata":{"trusted":true,"_uuid":"dbe46080dab73dc67f23c3f6d8b4bb7137c86e54"},"cell_type":"code","source":"model_rf = RandomForestClassifier(n_estimators = 100 , random_state = 99).fit(x_train , y_train)\ntest_pred_rf = model_rf.predict(x_test)\nprint(accuracy_score(y_test , test_pred_rf))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c6c0bc1d3871789af9c389f648971c1d1a647242"},"cell_type":"markdown","source":"## Gradient Boost Count Vectorizer"},{"metadata":{"trusted":true,"_uuid":"623e92ec08a22557be45244cb2e4fd3ca30eaef7"},"cell_type":"code","source":"model_gb = GradientBoostingClassifier(n_estimators = 100 , random_state = 99).fit(x_train , y_train)\ntest_pred_gb = model_gb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_gb))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3f06ae231c81007b6506158c92203b22aa53907f"},"cell_type":"markdown","source":"## TF-IDF Vectorizer"},{"metadata":{"trusted":true,"_uuid":"1e343d86c5d9ed7407825f386a12e93866a07dac"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2744db22da4283b74f23ddfe3849315f480dfe3a"},"cell_type":"code","source":"x_train , x_test , y_train , y_test = train_test_split(Docs_clean , df_hotstar['Sentiment_Manual'] , \n                                                       test_size = 0.2 , random_state = 100)\ntfidf = TfidfVectorizer(min_df = 5).fit(x_train)\nx_train = tfidf.transform(x_train)\nx_test = tfidf.transform(x_test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ed400d9b7cb26d785a1c004d88072a8ed26707f8"},"cell_type":"markdown","source":"## Multinomial Naive Bayes using TFIDF Vectorizer"},{"metadata":{"trusted":true,"_uuid":"2d1fb2eb05f4e5b4d800c19b6b756247eaa53d46"},"cell_type":"code","source":"model_mnb = MultinomialNB().fit(x_train , y_train)\ntest_pred_mnb = model_mnb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_mnb))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"895856983ba5bbe4212e7ddbd28f145226af45df"},"cell_type":"markdown","source":"## Word2Vec"},{"metadata":{"trusted":true,"_uuid":"acde29ea5a1f2135c6c8a606fc614ea5e4e542bd"},"cell_type":"code","source":"docs_vectors = pd.DataFrame() # document-Term Matrix\nfor doc in Docs_clean:\n    words = nltk.word_tokenize(doc)\n    temp = pd.DataFrame()\n    for word in words:\n        try:\n            word_vec = embeddings[word]\n            temp = temp.append(pd.Series(word_vec) , ignore_index = True)\n        except:\n            pass\n    docs_vectors = docs_vectors.append(temp.mean() , ignore_index = True)\ndocs_vectors.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4121503227de7dc39b2e91edfc78a5275345a20d"},"cell_type":"markdown","source":"## Null vectors identification"},{"metadata":{"trusted":true,"_uuid":"24930651dcd27611297f695944ca000101a6cf9a"},"cell_type":"code","source":"null_vec = pd.DataFrame(pd.isnull(docs_vectors).sum(axis = 1).sort_values(ascending = False))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b14ac63f8ccaaabedfb5c51f1f19fa20e77a021"},"cell_type":"code","source":"null_vec.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d442e292b87c16188fcc33a9489e47abcd6f4099"},"cell_type":"code","source":"nl = null_vec.index[null_vec[0]==300].tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b97ddc63499c745c8bcf1e26b65c6c68582b6d65"},"cell_type":"code","source":"len(nl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6d91928355f33cf92e013434e3d7660977b0fb4"},"cell_type":"code","source":"x = docs_vectors.drop(nl)\ny = df_hotstar['Sentiment_Manual'].drop(nl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4ce18b1e0b72309ba809a7234ec5aa13a40531d"},"cell_type":"code","source":"x.shape , y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec62fc056ee946b629a5b1411f8036a6778d14e0"},"cell_type":"code","source":"x_train , x_test , y_train , y_test = train_test_split(x , y , test_size = 0.2 , random_state = 100)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"19ae24378b8184249890ffe8561b9a75d3791a06"},"cell_type":"markdown","source":"## Random Forest Classifier Word2Vec"},{"metadata":{"trusted":true,"_uuid":"a245a0279e1d0a1865ae9660aafddaac98f169de"},"cell_type":"code","source":"model_rf = RandomForestClassifier(n_estimators = 100).fit(x_train , y_train)\ntest_pred_rf = model_rf.predict(x_test)\nprint(accuracy_score(y_test , test_pred_rf))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b27daa75e71540c021a1eaf3392d41cb275899e1"},"cell_type":"markdown","source":"## Ada Boost Classifier Word2Vec"},{"metadata":{"trusted":true,"_uuid":"68c10fd2f5670bd640b960761d458361a0065f60"},"cell_type":"code","source":"model_ab =AdaBoostClassifier(n_estimators = 100).fit(x_train , y_train)\ntest_pred_ab = model_ab.predict(x_test)\nprint(accuracy_score(y_test , test_pred_ab))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d0fb4a11936b63bfbfcf7e3ae4464cb80aa7118"},"cell_type":"markdown","source":"## Gradient Boost Classifier Word2Vec"},{"metadata":{"trusted":true,"_uuid":"75f32ec3f53d54c578380a6f39ec75edd372f559"},"cell_type":"code","source":"model_gb = GradientBoostingClassifier(n_estimators = 100).fit(x_train , y_train)\ntest_pred_gb = model_gb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_gb))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"923318350d4e41b1acdb6dd1fe1a816a19d89786"},"cell_type":"markdown","source":"## Sentiment Predicition using VADER"},{"metadata":{"trusted":true,"_uuid":"80371f861f719bfdec510cef87df905f4b86c1d0"},"cell_type":"code","source":"from nltk.sentiment import SentimentIntensityAnalyzer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a405f97d89895020e22eb271404e421c1dc54909"},"cell_type":"code","source":"analyzer = SentimentIntensityAnalyzer()\n\ndef get_sentiment (sentence , analyzer = analyzer):\n    compound = analyzer.polarity_scores(sentence)['compound']\n    if compound > 0.1:\n        return 'Positive'\n    elif compound < 0.1:\n        return 'Negative'\n    else:\n        return 'Neutral'    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66e5085df41e9fc17b566a3f72107a68c3cb0f8e"},"cell_type":"code","source":"df_hotstar = df_hotstar.drop(['Sentiment_Vader'] , axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89bed59543f85b5a34ced06f786ad3dcb6e0ad7b"},"cell_type":"code","source":"df_hotstar.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33afae728a8d043f266b02dbee4df4b06385fcf9"},"cell_type":"code","source":"df_hotstar['Sentiment_Vader'] = df_hotstar['Reviews'].apply(get_sentiment)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f7bf787b947855a486e2a1972986a5135edea22"},"cell_type":"code","source":"accuracy_score(df_hotstar['Sentiment_Manual'] , df_hotstar['Sentiment_Vader'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c7245c2e6340b45b614b054c46e3818a83868e8"},"cell_type":"code","source":"df_hotstar.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"452c0feffec647667d2e142bffd096e095ff735f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}