{"cells":[{"metadata":{"_uuid":"a71fd15f32bd5868c0d8c2ad0b2cbace7580e642"},"cell_type":"markdown","source":"Word2Vec for IMDB ratings"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gensim\nimport nltk\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier , AdaBoostClassifier\nfrom sklearn.naive_bayes import GaussianNB , BernoulliNB\nfrom sklearn.metrics import accuracy_score\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300/\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"path = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(path , binary = True)\n# Collection of words are listed in embeddings\n# Default is 8 GB, shorter version will be provided\n# Black box model\n# For every word there are 300 lists","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d18ad9524b17a08d3f2d41fce5a9b0a4ae423f93"},"cell_type":"code","source":"embeddings['amazon']\nlen(embeddings['amazon'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6def647586f66ce5f61687a7363b9be07a2b43ab"},"cell_type":"code","source":"embeddings.most_similar('rahul' , topn = 10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f826630bd68afa01041b013495d066011ce4924"},"cell_type":"code","source":"embeddings.most_similar('hyundai' , topn = 10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa8ac152196b7cd4d983bf2103b77f14b0126155"},"cell_type":"code","source":"embeddings.doesnt_match(['football' , 'basketball' , 'cricket' , 'apple'])\n# Cosine similarity among football , basketball , cricket are very high hence apple is given as odd man out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5764c577e7926d3fb4c72c67d276dcf18996e52"},"cell_type":"code","source":"url = 'https://raw.githubusercontent.com/skathirmani/datasets/master/imdb_sentiment.csv'\ndf_imdb = pd.read_csv(url)\ndf_imdb['review'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5b77900a72f3a450e130b3680e20f3543012adb"},"cell_type":"code","source":"# one option is to add vectors with their elements\n# Other option is to take average\n# weights are the embeddings and are calculated in deep learning\n#for the the 1st document a temporary df is created: first convert the document to a word-weight matrix and then compute column average\n# the temporary df is created for all the documents in the corpus\n# the final df contains the column weights for all the documents\n# the # of columns are always 300 in the final df\n# the #of rows depends on the input","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afa789def319530c971d0676ece81e23239d265f"},"cell_type":"code","source":"df_imdb.loc[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eddd14ebd67596ab42f2029cc84707cf942a35e5"},"cell_type":"code","source":"doc = df_imdb.loc[0 , 'review']\nwords = nltk.word_tokenize(doc.lower())\ntemp = pd.DataFrame()\nfor word in words:\n    try:\n        print(embeddings[word][:5])\n        temp = temp.append(pd.Series(embeddings[word]) , ignore_index = True)\n    except:\n        print(word, 'is not there')\ntemp\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26d6f5abe4fac5fef1383323d817464214a2ec7b"},"cell_type":"code","source":"temp.mean()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"94b6ca35e9eb397ebfa66a18719df9cc51ca9001"},"cell_type":"markdown","source":"## Data Cleaning"},{"metadata":{"trusted":true,"_uuid":"d192203e76f34e54e31a37f81ac17a2f609b39df"},"cell_type":"code","source":"docs = df_imdb['review'].str.replace('-' , ' ').str.lower().str.replace('[^a-z ]' , '')\nStopWords = nltk.corpus.stopwords.words('english')\nclean_sentence = lambda doc: ' '.join([word for word in nltk.word_tokenize(doc) if word not in StopWords])\ndocs_clean = docs.apply(clean_sentence)\ndocs_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"feae1b2163342dc6d024c37ef4fd548afbb98147"},"cell_type":"code","source":"# Final DF\ndocs_vectors = pd.DataFrame() # document-Term Matrix\nfor doc in docs_clean:\n    words = nltk.word_tokenize(doc)\n    temp = pd.DataFrame()\n    for word in words:\n        try:\n            word_vec = embeddings[word]\n            temp = temp.append(pd.Series(word_vec) , ignore_index = True)\n        except:\n            pass\n    docs_vectors = docs_vectors.append(temp.mean() , ignore_index = True)\ndocs_vectors.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aeddef0ad6893276874dd77139080114d0448113"},"cell_type":"code","source":"pd.isnull(docs_vectors).sum(axis = 1).sort_values(ascending = False)\n# In 64 and 590th row, there are missing values ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b8c75534932c40b536a4c2583ae709360fd13c2"},"cell_type":"code","source":"df_imdb.loc[64 , 'review']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3a6fc96f4c4fc8c2a7eca0635b0151e7cb96071"},"cell_type":"code","source":"df_imdb.loc[590 , 'review']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6da9f8e1ccd5a990e64a667cbcf5d713db847bf0"},"cell_type":"code","source":"# since 64th row and 590th row are numbers we are dropping those rows\nx = docs_vectors.drop([64 , 590])\ny = df_imdb['sentiment'].drop([64 , 590])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65f27bfa636d32dad83c3f648b57772f96ebf63e"},"cell_type":"code","source":"x_train , x_test , y_train , y_test = train_test_split(x , y , test_size = 0.2 , random_state = 100)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7388eb808fd2e83a2afc8991f534379270266fd4"},"cell_type":"markdown","source":"> ## Random Forest"},{"metadata":{"trusted":true,"_uuid":"a35e93ebad93963550491321a185bd734904f6ee"},"cell_type":"code","source":"model_rf = RandomForestClassifier(n_estimators = 800).fit(x_train , y_train)\ntest_pred_rf = model_rf.predict(x_test)\nprint(accuracy_score(y_test , test_pred_rf))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7987395d9e6208d3cf45e43a0d446900618e4e9"},"cell_type":"markdown","source":"## Ada Boost Classifier"},{"metadata":{"trusted":true,"_uuid":"ea57edb28c48c93d20089df071a2f24c5a1aa29a"},"cell_type":"code","source":"model_ab =AdaBoostClassifier(n_estimators = 800).fit(x_train , y_train)\ntest_pred_ab = model_ab.predict(x_test)\nprint(accuracy_score(y_test , test_pred_ab))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9c372629120fdf36c9960ce14ea91857b11f4c64"},"cell_type":"markdown","source":"## Gaussian Naive Bayes Classifier"},{"metadata":{"trusted":true,"_uuid":"51e2b16793f638e9ee1785c30259cbda5deefacc"},"cell_type":"code","source":"model_gnb = GaussianNB().fit(x_train , y_train)\ntest_pred_gnb = model_gnb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_gnb))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"56cf6a1ba16612d0e130e73e3b961e498e69e605"},"cell_type":"markdown","source":"## Bernoulli Naive Bayes Classifier"},{"metadata":{"trusted":true,"_uuid":"be8d9e3f792957f852d6555f01a6bfd57fcc3d7c"},"cell_type":"code","source":"model_bnb = BernoulliNB().fit(x_train , y_train)\ntest_pred_bnb = model_bnb.predict(x_test)\nprint(accuracy_score(y_test , test_pred_bnb))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74842a8995b4c97d5eec4b50a4da985adc9aa2f3"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}