{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\n\nimport gensim\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))     \n### ^^^^***** under 'embeddings' -> GoogleNews, wiki-news... these folders to use for extracting files (.bin)\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"url = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\n\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(url, binary = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"41031c31d5620256cbe577a41d0e78169e67bc71"},"cell_type":"code","source":"embeddings['sabermetrics'] ## presence of word in news","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"bd94ae1468b4b9652478e1eec64ed7ebe2f43c7c"},"cell_type":"code","source":"embeddings['ghuiya'] ## NOT present in news; hence, ERROR","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d096488fbdd7ee0115f071a4df15bade0eeacafc"},"cell_type":"code","source":"embeddings.most_similar('camera', topn = 10)   ## based on Cosine Similarity , to find similar terms","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7522801513e03c1c1345bf65733cddcf261c0e9b"},"cell_type":"code","source":"embeddings.doesnt_match(['king', 'woman', 'gandhi', 'sonia'])  ## odd one out \n\n## king - man = queen","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c6815af7af43d46a8b30374d08da4c678c9724b"},"cell_type":"code","source":"embeddings.most_similar(positive = ['king', 'woman'], negative = ['man'], topn = 10) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"363f7aa5dcd7b7bdc41fb676df82caf3a3b99a0d"},"cell_type":"code","source":"url = 'https://bit.ly/2S2yXEd'  ## from the Internet, downloading a link for text-classification\nimdb = pd.read_csv(url)\nimdb.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27892bcbef68a40b12ca2b83cff1c247b6fbf2a5"},"cell_type":"code","source":"imdb.loc[0, 'review']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"548cfe791521de131274a11efff72a61d4d08eda","scrolled":true},"cell_type":"code","source":"embeddings['A'] \n\n### ^^^^^^ checking the embedding value, for which the Gradient Descent formula is used to calculate weights ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ede66521e95024019d061978b2b2364fa54770df"},"cell_type":"code","source":"''' ## without stopwords ##\nimport nltk \n\ndocs_vectors = pd.DataFrame()\n\n## in below... all lowercase shall help in covering all the words, instead of adding \"\"A-Z\"\" in RegEx which may not provide suitable outputs\nfor doc in imdb['review'].str.lower().str.replace('[^a-z ]', ''):\n    temp = pd.DataFrame()   ## initially empty, and empty on every iteration\n    for word in nltk.word_tokenize(doc):  ## choose one word at a time from the doc from above\n        try:\n            word_vec = embeddings[word]  ## if present, the following code applies\n            temp = temp.append(pd.Series(word_vec), ignore_index = True)  ## .Series to make it easier to append \"without\" index labels\n        except:\n            pass\n    doc_vector = temp.mean()\n    docs_vectors = docs_vectors.append(doc_vector, ignore_index = True) ## added to the empty data frame\ndocs_vectors.shape ## 300 columns is a lot lesser'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9c2dedf6b381c0d385f5172a1f4063fc6dda57c"},"cell_type":"code","source":"## with stopwords ##\n\nimport nltk\n\ndocs_vectors = pd.DataFrame()\nstopwords = nltk.corpus.stopwords.words('english')   ## !! added later\n\n## in below... all lowercase shall help in covering all the words, instead of adding \"\"A-Z\"\" in RegEx which may not provide suitable outputs\nfor doc in imdb['review'].str.lower().str.replace('[^a-z ]', ''):\n    temp = pd.DataFrame()   ## initially empty, and empty on every iteration\n    for word in doc.split(' '):  ## !!\n        if word not in stopwords: \n            try:\n                word_vec = embeddings[word]  ## if present, the following code applies\n                temp = temp.append(pd.Series(word_vec), ignore_index = True)  ## .Series to make it easier to append \"without\" index labels\n            except:\n                pass\n    doc_vector = temp.mean()\n    docs_vectors = docs_vectors.append(doc_vector, ignore_index = True) ## added to the empty data frame\ndocs_vectors.shape ## 300 columns is a lot lesser","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10e937ffe8d194c46f63f06fb861458b0a9df6f7"},"cell_type":"code","source":"docs_vectors.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"faf0a78d8eedba22516bc39cc2eccff8d9d7c98a"},"cell_type":"code","source":"pd.isnull(docs_vectors).sum().sum() ## 600 nulls, when all in lowercase","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e261cee615fd47a996d24be55a7cc6ecb4886a4"},"cell_type":"code","source":"## adding a column in docs_vector of \"sentiment\"  + dropping the null values\ndocs_vectors['sentiment'] = imdb['sentiment']\ndocs_vectors = docs_vectors.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6bf346e972b3236694f1c591d3c55ec0255ac104"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split ## here vectorization (vectorizer) again shall not come, since we are calculated weights \nfrom sklearn.ensemble import AdaBoostClassifier\n\ntrain_x, test_x, train_y, test_y = train_test_split(docs_vectors.drop('sentiment', axis = 1),\n                                                   docs_vectors['sentiment'],\n                                                   test_size = 0.2,\n                                                   random_state = 1)\ntrain_x.shape, test_x.shape, train_y.shape, test_y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67c4b69acc27aa1a97a41d25482cb48c5817227e"},"cell_type":"code","source":"model = AdaBoostClassifier(n_estimators = 800, random_state = 1)\nmodel.fit(train_x, train_y)\ntest_pred = model.predict(test_x)\n\nfrom sklearn.metrics import accuracy_score\naccuracy_score(test_y, test_pred)  \n\n## 74% accuracy without stopwords \n\n## 75.33% accuracy with stopwords","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f9f7f31dcac1745ffccd60ac76db64d8376d772e"},"cell_type":"code","source":"'''model = AdaBoostClassifier(n_estimators = 2000, random_state = 5)\nmodel.fit(train_x, train_y)\ntest_pred = model.predict(test_x)\n\nfrom sklearn.metrics import accuracy_score\naccuracy_score(test_y, test_pred)  \n\n## 74% accuracy without stopwords \n\n## 75.33% accuracy with stopwords''' ## accuracy fell to 74.6% with stopwords","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a9aaf7d8f47a9bec390eecf0d5c176309f75431"},"cell_type":"markdown","source":"## VADER package :\n\nValence Aware Dictionary and sEntiment Reasoner"},{"metadata":{"_uuid":"ae9f5c2bb412fc3afcf25291fb4472086eeb1267"},"cell_type":"markdown","source":"NOTE:\n'''presence of punctuations, capitals make impact on the individual / overall score... so DO NOT clean data or change anything in the text to avoid distorting the score\n\nAlso, works well for shorter documents.\n\nSTOPWORDS shall be HEEDED!\n\nSingle letter words are IGNORED! '''"},{"metadata":{"trusted":true,"_uuid":"c5866925d73180bf1f5ce0cb4609079a07daca3f"},"cell_type":"code","source":"from nltk.sentiment import SentimentIntensityAnalyzer\n\nsentiment = SentimentIntensityAnalyzer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"046cb6f663fee7f1c588b5cdb50f120932f33392"},"cell_type":"code","source":"reviews = imdb['review'].str.lower().str.replace('[^a-z ]', '')\nreviews","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d647a6a78ec5f8257380b17200745c7fa9158deb"},"cell_type":"markdown","source":"#### FORMULA:\ncompound score = [score / sqrt{(score^2)+alpha}]****"},{"metadata":{"trusted":true,"_uuid":"f5cdb17d3bf95fcae7f75fb468b0e6fe52f936eb"},"cell_type":"code","source":"imdb['sentiment'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"c7b19cfea1e22cb0329d4c46a1135cce51d62d60"},"cell_type":"code","source":"def get_sentiment(text):\n    sentiment = SentimentIntensityAnalyzer() #### calling Intensity Analyzer\n    compound = sentiment.polarity_scores(text)['compound']  ### calling the 'compound' score for the \"text\" entered\n    if compound > 0:\n        return 1  ## positive\n    else:\n        return 0 ## negative\n    #else:\n        #return \"Neutral\"     \n    return compound\n\nimdb['sentiment_vader'] = imdb['review'].apply(get_sentiment) ### in the columns of \"imdb\"\nimdb['sentiment_vader'] ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc4960b2222c1ab525b175968df3f0d904bbf18e"},"cell_type":"code","source":"get_sentiment(\"YES\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"89daa8cc8c9a3b45281ece270144d5aad09dc34a"},"cell_type":"markdown","source":"### Calculating Accuracy Score using VADER"},{"metadata":{"trusted":true,"_uuid":"b3472f97f1a561458c65599b1aea00853ff6df81"},"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\naccuracy_score(imdb['sentiment'], imdb['sentiment_vader']) ## == 79.011% of accuracy score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff193fe054def953623f071feb573bf4eb6a1d89"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}