{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport os\n\nimport gensim\n\nprint(os.listdir(\"../input/quora-insincere-questions-classification/embeddings/GoogleNews-vectors-negative300/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f0aacbf314c6d7cb15e19b657d352aa15537b88"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36209e509d1a1792b20b57dd16a714e7205c1237"},"cell_type":"code","source":"link = \"../input/quora-insincere-questions-classification/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\n\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(link, binary = True)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"## Reading files\n\nyelp = pd.read_csv(\"../input/yelp-sentiment-dataset/yelp.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d88817de3fd0816079bbc9203345842b09ad78f7"},"cell_type":"code","source":"yelp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"085e81c0807941ab3c91fab6e48290dc11305d9c"},"cell_type":"code","source":"yelp = yelp.drop(\"Unnamed: 0\", axis=1)\nyelp.columns   ## Dropping the not-utilisable column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87648d9e63a1d67022026ef602ccfdcf76aab9ff"},"cell_type":"code","source":"yelp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6c845fabc3c1f943b828e52e2b79f8370093bfc"},"cell_type":"code","source":"## Using Stopwords (i.e removed)##\n\nimport nltk\n\ndocs_vectors = pd.DataFrame()  ## empty dataframe\nstopwords = nltk.corpus.stopwords.words('english')   ## !! added later\n\n## in below... all lowercase shall help in covering all the words, instead of adding \"\"A-Z\"\" in RegEx which may not provide suitable outputs\nfor doc in yelp['review'].str.lower().str.replace('[^a-z ]', ''):\n    temp = pd.DataFrame()   ## initially empty, and empty on every iteration\n    for word in doc.split(' '):  ## !!\n        if word not in stopwords: \n            try:\n                word_vec = embeddings[word]  ## if present, the following code applies\n                temp = temp.append(pd.Series(word_vec), ignore_index = True)  ## .Series to make it easier to append \"without\" index labels\n            except:\n                pass\n    doc_vector = temp.mean()\n    docs_vectors = docs_vectors.append(doc_vector, ignore_index = True) ## added to the empty data frame\n\n# docs_vectors.shape ## ==> (1000 x 300) order","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65d5b07d7bff364ba76ad8212bef5b72d030a1b8"},"cell_type":"code","source":"docs_vectors.head() ## a sparse matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6edf8d72be14471e19150758453e910d8a5bd367"},"cell_type":"code","source":"pd.isnull(docs_vectors).sum().sum() # No null values present","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8597667eb487de327cf1a7966b24bd55abb3070f"},"cell_type":"code","source":"## adding a column in docs_vector of \"sentiment\"  + dropping the null values\n\ndocs_vectors['sentiment'] = yelp['sentiment']\ndocs_vectors = docs_vectors.dropna()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dea8674c004d10780872fd7adbaf47d30782760d"},"cell_type":"markdown","source":"### Adaptive Boost algorithm \n    - ****(to check accuracy of predictions on Yelp reviews)"},{"metadata":{"trusted":true,"_uuid":"76fac553be4f2c58e40af57fb0428e4faddf8957"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split \n\n## here vectorization (vectorizer) again shall not come, since we are calculated weights \nfrom sklearn.ensemble import AdaBoostClassifier \n\ntrain_x, test_x, train_y, test_y = train_test_split(docs_vectors.drop('sentiment', axis = 1),\n                                                   docs_vectors['sentiment'],\n                                                   test_size = 0.2,\n                                                   random_state = 1)\n\ntrain_x.shape, test_x.shape, train_y.shape, test_y.shape  ## Test and Train partitions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de85c80240c20bbc7e7485404b1ddf1bdc366ca4"},"cell_type":"code","source":"model = AdaBoostClassifier(n_estimators = 900, random_state = 1)\nmodel.fit(train_x, train_y)\n\ntest_pred = model.predict(test_x)\n\nfrom sklearn.metrics import accuracy_score\naccuracy_score(test_y, test_pred)   \n\n## == 77.5% accuracy score using AdaBoost algorithm (with Stopwords removed)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa6ebd86ff484942c284e7cbc534dcf7fa1dfa99"},"cell_type":"markdown","source":"## **VADER package :\nValence Aware Dictionary and sEntiment Reasoner\n\nNOTE: '''presence of punctuations, capitals make impact on the individual / overall score... so DO NOT clean data or change anything in the text to avoid distorting the score\n\nAlso, works well for shorter documents.\n\nSTOPWORDS shall be HEEDED!\n\nSingle letter words are IGNORED! '''"},{"metadata":{"trusted":true,"_uuid":"d7b2bc16b008d4f084198845c723bd80fdcd616f"},"cell_type":"code","source":"### Sentiment Analyzer to check out Sentiments\n\nfrom nltk.sentiment import SentimentIntensityAnalyzer\n\nsentiment = SentimentIntensityAnalyzer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fce96d47335a2c8ae4559d654325798665f4efa4"},"cell_type":"code","source":"reviews = yelp['review'].str.lower().str.replace('[^a-z ]', '')\nreviews   \n\n## the Yelp reviews are put into lowercase and then using RegEx, words are split seperated.\n## this format allows for better analysis of sentiment of words","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8dd2728fab41d58304480224626582cbc12cbceb"},"cell_type":"markdown","source":"#### NOTE:\nIn \"compound score\" is taken from the following calculations are made - \n\ncompound score = [score / sqrt{(score^2)+alpha}]"},{"metadata":{"trusted":true,"_uuid":"e65da3e0ed145e859488556f383de30883bc25d0"},"cell_type":"code","source":"yelp['sentiment'].value_counts()   ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"321b0deb2a0dd9e7ae6250d6a9e5d3f6ef57f09c"},"cell_type":"code","source":"## Using a user-defined function to find out the sentiment out of Yelp reviews\n\ndef get_sentiment(text):\n    sentiment = SentimentIntensityAnalyzer() #### calling Intensity Analyzer\n    compound = sentiment.polarity_scores(text)['compound']  ### calling the 'compound' score for the \"text\" entered\n    if compound > 0:\n        return 1  ## positive\n    else:\n        return 0 ## negative\n    #else:\n        #return \"Neutral\"     \n    return compound\n\nyelp['sentiment_vader'] = yelp['review'].apply(get_sentiment) ### in the columns of \"imdb\"\nyelp['sentiment_vader'] ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7d80ed1a915f4af818134586d442db603a24eb88"},"cell_type":"markdown","source":"#### Calculating accuracy score using VADER"},{"metadata":{"trusted":true,"_uuid":"878433f2aba6765a3e64ad3bfd0c9a94a646d84c"},"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\naccuracy_score(yelp['sentiment'], yelp['sentiment_vader']) ## == 80.9% of accuracy score using VADER\n\n## ==> improved accuracy using VADER","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d7f57aa15e3cab3b7b0df749e5e45ffc7003712"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}