{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gensim\n#print(os.listdir(\"../input\"))\nprint(os.listdir('../input/embeddings/GoogleNews-vectors-negative300/'))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"path = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(path,binary = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16ea58f9a52a2941a1235755ad2c5bd03a62bff1"},"cell_type":"markdown","source":"## collection of word vectors is called ** word embeddings **  "},{"metadata":{"trusted":true,"_uuid":"4ad8e161bcc68a60ff8eeb0e6d963f690594a050"},"cell_type":"code","source":"len(embeddings['rahul'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed3f992d9c6d9bfa552db07a13297a127f11266b"},"cell_type":"code","source":"embeddings.most_similar('rahul',topn = 10) #top 10 words similar to rahul","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7cf6ba2d56d107d4654b5db8ecb09b446d2fbda5"},"cell_type":"code","source":"embeddings.doesnt_match(['football','basketball','cricket','apple'])\n#Cosine similarity is checked. Apple has the least wtr to other","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"81862dd8b51284a3c7d63daa934b6e3e5e7d8765"},"cell_type":"code","source":"url ='https://bit.ly/2S2yXEd'\nimdb = pd.read_csv(url)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eebdd9c132626caf96175f993b41d37fa8d725f0"},"cell_type":"code","source":"imdb['review'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5034e115aecca57de423a8c74897e52cb8782cc8"},"cell_type":"code","source":"import nltk\ndoc = imdb.loc[0,'review']\nwords = nltk.word_tokenize(doc.lower())\n\ntemp = pd.DataFrame()\nfor word in words:\n    try:\n        print(embeddings[word][:5])\n        temp = temp.append(pd.Series(embeddings[word]),ignore_index= True)\n        #temp\n    except:\n        print(word,'does not have a vector representation')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"620e585f50071b7867ea9e0898b8f030283ae7d1"},"cell_type":"code","source":"docs = imdb['review'].str.replace('-',' ').str.lower().str.replace('[^a-z ]','') \nstopwords = nltk.corpus.stopwords.words('english')\nclean_sentence = lambda doc: ' '.join([word for word in nltk.word_tokenize(doc) if word not in stopwords])\ndocs_clean = docs.apply(clean_sentence)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df15a152a079582547d452cd2952d0b2e0dd6ca7"},"cell_type":"code","source":"docs_vectors = pd.DataFrame()\nfor doc in docs_clean:\n    words = nltk.word_tokenize(doc)\n    temp = pd.DataFrame()\n    for word in words : \n        try: \n            word_vec = embeddings[word]\n            temp = temp.append(pd.Series(word_vec), ignore_index= True)\n        except:\n            pass #goes to the next word \n    docs_vectors = docs_vectors.append(temp.mean(), ignore_index = True)\ndocs_vectors.shape\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1657cd67e6a4feb1146993000df2a5cbe2dc4dd2"},"cell_type":"code","source":"docs_vectors.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3748203546334247f6151bd20a368ba19a1de8c"},"cell_type":"code","source":"X = docs_vectors.drop([64,590])\ny = imdb['sentiment'].drop([64,590])\n\nfrom sklearn.model_selection import train_test_split\ntrain_x , test_x , train_y, test_y = train_test_split(X,y,\n                                                     test_size = 0.2 , random_state = 100 )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"088abfa1d0e01c14181f48583f509265724bbd0b"},"cell_type":"markdown","source":"# Random Forest Classifier"},{"metadata":{"trusted":true,"_uuid":"7700bc679111a37bbc5bdeaecba3357554629c3c"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier,AdaBoostClassifier \nfrom sklearn.metrics import accuracy_score\n\nmodel_ran = RandomForestClassifier(n_estimators = 300 ).fit(train_x,train_y)\ntest_pred = model_ran.predict(test_x)\naccuracy_score(test_y,test_pred)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e0f2b6d8a93a5c28013b1c2c0325d9f1db152e22"},"cell_type":"markdown","source":"# Adaboost Classifier"},{"metadata":{"trusted":true,"_uuid":"121599ce005ca350900e970c43333ba156b563ea"},"cell_type":"code","source":"model_ada = AdaBoostClassifier(n_estimators= 800).fit(train_x , train_y)\ntest_pred_ada = model_ada.predict(test_x)\naccuracy_score(test_y,test_pred_ada)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"09f78b308feb99c8c54669035e240b4c12c205d0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e5218ebf17a0a0e2ba9b49b9e11c9af9d23bab9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}