{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport nltk\nimport gensim\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[{"output_type":"stream","text":"['GoogleNews-vectors-negative300.bin']\n","name":"stdout"}]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"url = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(url, binary=True)","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cab2127b17304fc720d2d04a75da78acd019d47"},"cell_type":"code","source":"#embeddings.most_similar('camera', topn = 5)\n#embeddings.doesnt_match(['apple','banana','flower'])\n#embeddings.most_similar(positive = ['king','woman'], negative = ['man'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2473ed01569b721987f1c90090e937c2f87880eb"},"cell_type":"code","source":"url = 'https://bit.ly/2CdYYuf'\nyelp = pd.read_csv(url, sep='\\t', header = None)\nyelp.head()","execution_count":69,"outputs":[{"output_type":"execute_result","execution_count":69,"data":{"text/plain":"                                                   0  1\n0                           Wow... Loved this place.  1\n1                                 Crust is not good.  0\n2          Not tasty and the texture was just nasty.  0\n3  Stopped by during the late May bank holiday of...  1\n4  The selection on the menu was great and so wer...  1","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>0</th>\n      <th>1</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>Wow... Loved this place.</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>Crust is not good.</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>Not tasty and the texture was just nasty.</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>Stopped by during the late May bank holiday of...</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>The selection on the menu was great and so wer...</td>\n      <td>1</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"yelp.rename(columns={0:'Reviews', 1:'Sentiment'}, inplace=True)\nyelp.head()","execution_count":70,"outputs":[{"output_type":"execute_result","execution_count":70,"data":{"text/plain":"                                             Reviews  Sentiment\n0                           Wow... Loved this place.          1\n1                                 Crust is not good.          0\n2          Not tasty and the texture was just nasty.          0\n3  Stopped by during the late May bank holiday of...          1\n4  The selection on the menu was great and so wer...          1","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>Reviews</th>\n      <th>Sentiment</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>Wow... Loved this place.</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>Crust is not good.</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>Not tasty and the texture was just nasty.</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>Stopped by during the late May bank holiday of...</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>The selection on the menu was great and so wer...</td>\n      <td>1</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"metadata":{"trusted":true,"_uuid":"b9e7580a07c762a8aaee1e8ac6ab21197e9ca3c6"},"cell_type":"code","source":"yelp.loc[0, 'Reviews']","execution_count":71,"outputs":[{"output_type":"execute_result","execution_count":71,"data":{"text/plain":"'Wow... Loved this place.'"},"metadata":{}}]},{"metadata":{"trusted":true,"_uuid":"e3cda5a675c9d7c57f9f28ecf2e3efe81e350696"},"cell_type":"code","source":"\ndocs_vectors = pd.DataFrame() # creating empty final dataframe\nstopwords = nltk.corpus.stopwords.words('english') # removing stop words\nfor doc in yelp['Reviews'].str.lower().str.replace('[^a-z ]', ''): # looping through each document and cleaning it\n    temp = pd.DataFrame()  # creating a temporary dataframe(store value for 1st doc & for 2nd doc remove the details of 1st & proced through 2nd and so on..)\n    for word in doc.split(' '): # looping through each word of a single document and spliting through space\n        if word not in stopwords: # if word is not present in stopwords then (try)\n            try:\n                word_vec = embeddings[word] # if word is present in embeddings(goole provides weights associate with words(300)) then proceed\n                temp = temp.append(pd.Series(word_vec), ignore_index = True) # if word is present then append it to temporary dataframe\n            except:\n                pass\n    doc_vector = temp.mean() # take the average of each column(w0, w1, w2,........w300)\n    docs_vectors = docs_vectors.append(doc_vector, ignore_index = True) # append each document value to the final dataframe\ndocs_vectors.shape","execution_count":72,"outputs":[{"output_type":"execute_result","execution_count":72,"data":{"text/plain":"(1000, 300)"},"metadata":{}}]},{"metadata":{"trusted":true,"_uuid":"a9adfa871164394746cc1401cfab7c03d424e70c"},"cell_type":"code","source":"pd.isnull(docs_vectors).sum().sum()","execution_count":73,"outputs":[{"output_type":"execute_result","execution_count":73,"data":{"text/plain":"0"},"metadata":{}}]},{"metadata":{"trusted":true,"_uuid":"369180a8ee4ec028611697a70904baa8d20ba1b9"},"cell_type":"code","source":"docs_vectors['Sentiment'] = yelp['Sentiment']\ndocs_vectors = docs_vectors.dropna()","execution_count":74,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41c704826976ccb3915f11e3c2bb3110c89b5d05"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import AdaBoostClassifier\n\ntrain_x, test_x, train_y, test_y = train_test_split(docs_vectors.drop('Sentiment', axis = 1),\n                                                   docs_vectors['Sentiment'],\n                                                   test_size = 0.2,\n                                                   random_state = 1)\ntrain_x.shape, train_y.shape, test_x.shape, test_y.shape","execution_count":75,"outputs":[{"output_type":"execute_result","execution_count":75,"data":{"text/plain":"((800, 300), (800,), (200, 300), (200,))"},"metadata":{}}]},{"metadata":{"trusted":true,"_uuid":"f6b3d50f90a9b5119e0777a02d7737f8a3e33ab7"},"cell_type":"code","source":"model = AdaBoostClassifier(n_estimators=800, random_state = 1)\nmodel.fit(train_x, train_y)\ntest_pred = model.predict(test_x)\nfrom sklearn.metrics import accuracy_score\naccuracy_score(test_y, test_pred)","execution_count":76,"outputs":[{"output_type":"execute_result","execution_count":76,"data":{"text/plain":"0.79"},"metadata":{}}]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}