{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport nltk\nimport gensim\n!pip install wget\n\n!wget -c \"https://s3.amazonaws.com/dl4j-distribution/GoogleNews-vectors-negative300.bin.gz\"\n\n# print(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-27T12:05:45.652838Z","iopub.execute_input":"2021-05-27T12:05:45.65316Z","iopub.status.idle":"2021-05-27T12:06:13.377815Z","shell.execute_reply.started":"2021-05-27T12:05:45.653108Z","shell.execute_reply":"2021-05-27T12:06:13.377026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !gzip -d ../input/embeddings.zip\n# url = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format('../output/kaggle/working/GoogleNews-vectors-negative300.bin', binary=True)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2021-05-27T12:11:04.122819Z","iopub.execute_input":"2021-05-27T12:11:04.12314Z","iopub.status.idle":"2021-05-27T12:11:04.142507Z","shell.execute_reply.started":"2021-05-27T12:11:04.123075Z","shell.execute_reply":"2021-05-27T12:11:04.141506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#embeddings.most_similar('camera', topn = 5)\n#embeddings.doesnt_match(['apple','banana','flower'])\n#embeddings.most_similar(positive = ['king','woman'], negative = ['man'])","metadata":{"_uuid":"3cab2127b17304fc720d2d04a75da78acd019d47","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"url = 'https://bit.ly/2CdYYuf'\nyelp = pd.read_csv(url, sep='\\t', header = None)\nyelp.head()","metadata":{"_uuid":"2473ed01569b721987f1c90090e937c2f87880eb","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yelp.rename(columns={0:'Reviews', 1:'Sentiment'}, inplace=True)\nyelp.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yelp.loc[0, 'Reviews']","metadata":{"_uuid":"b9e7580a07c762a8aaee1e8ac6ab21197e9ca3c6","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndocs_vectors = pd.DataFrame() # creating empty final dataframe\nstopwords = nltk.corpus.stopwords.words('english') # removing stop words\nfor doc in yelp['Reviews'].str.lower().str.replace('[^a-z ]', ''): # looping through each document and cleaning it\n    temp = pd.DataFrame()  # creating a temporary dataframe(store value for 1st doc & for 2nd doc remove the details of 1st & proced through 2nd and so on..)\n    for word in doc.split(' '): # looping through each word of a single document and spliting through space\n        if word not in stopwords: # if word is not present in stopwords then (try)\n            try:\n                word_vec = embeddings[word] # if word is present in embeddings(goole provides weights associate with words(300)) then proceed\n                temp = temp.append(pd.Series(word_vec), ignore_index = True) # if word is present then append it to temporary dataframe\n            except:\n                pass\n    doc_vector = temp.mean() # take the average of each column(w0, w1, w2,........w300)\n    docs_vectors = docs_vectors.append(doc_vector, ignore_index = True) # append each document value to the final dataframe\ndocs_vectors.shape","metadata":{"_uuid":"e3cda5a675c9d7c57f9f28ecf2e3efe81e350696","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.isnull(docs_vectors).sum().sum()","metadata":{"_uuid":"a9adfa871164394746cc1401cfab7c03d424e70c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"docs_vectors['Sentiment'] = yelp['Sentiment']\ndocs_vectors = docs_vectors.dropna()","metadata":{"_uuid":"369180a8ee4ec028611697a70904baa8d20ba1b9","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import AdaBoostClassifier\n\ntrain_x, test_x, train_y, test_y = train_test_split(docs_vectors.drop('Sentiment', axis = 1),\n                                                   docs_vectors['Sentiment'],\n                                                   test_size = 0.2,\n                                                   random_state = 1)\ntrain_x.shape, train_y.shape, test_x.shape, test_y.shape","metadata":{"_uuid":"41c704826976ccb3915f11e3c2bb3110c89b5d05","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AdaBoostClassifier(n_estimators=800, random_state = 1)\nmodel.fit(train_x, train_y)\ntest_pred = model.predict(test_x)\nfrom sklearn.metrics import accuracy_score\naccuracy_score(test_y, test_pred)","metadata":{"_uuid":"f6b3d50f90a9b5119e0777a02d7737f8a3e33ab7","trusted":true},"execution_count":null,"outputs":[]}]}