{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\n\n\nimport os\nimport re\nimport nltk\nimport gensim\nimport string","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-08T18:47:23.404308Z","iopub.execute_input":"2022-05-08T18:47:23.404706Z","iopub.status.idle":"2022-05-08T18:47:23.410505Z","shell.execute_reply.started":"2022-05-08T18:47:23.404640Z","shell.execute_reply":"2022-05-08T18:47:23.409699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load pre-trained Word2Vec by Google\nFind it [here](https://code.google.com/archive/p/word2vec/)","metadata":{}},{"cell_type":"code","source":"url = \"../input/googlenewsvectorsnegative300/GoogleNews-vectors-negative300.bin\"\nembeddings = gensim.models.KeyedVectors.load_word2vec_format(url, binary=True)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2022-05-08T16:29:31.981868Z","iopub.execute_input":"2022-05-08T16:29:31.982203Z","iopub.status.idle":"2022-05-08T16:31:08.214281Z","shell.execute_reply.started":"2022-05-08T16:29:31.982149Z","shell.execute_reply":"2022-05-08T16:31:08.213504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Embeddings type:\" ,type(embeddings.vectors))\nprint(\"Embeddings shape: \", embeddings.vectors.shape)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T16:34:23.418923Z","iopub.execute_input":"2022-05-08T16:34:23.419203Z","iopub.status.idle":"2022-05-08T16:34:23.426119Z","shell.execute_reply.started":"2022-05-08T16:34:23.419153Z","shell.execute_reply":"2022-05-08T16:34:23.425039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings.vectors","metadata":{"execution":{"iopub.status.busy":"2022-05-08T17:04:21.834295Z","iopub.execute_input":"2022-05-08T17:04:21.834582Z","iopub.status.idle":"2022-05-08T17:04:21.844055Z","shell.execute_reply.started":"2022-05-08T17:04:21.834533Z","shell.execute_reply":"2022-05-08T17:04:21.843207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings.most_similar('camera', topn = 5)\n#embeddings.doesnt_match(['apple','banana','flower'])\n#embeddings.most_similar(positive = ['king','woman'], negative = ['man'])","metadata":{"_uuid":"3cab2127b17304fc720d2d04a75da78acd019d47","execution":{"iopub.status.busy":"2022-05-08T16:34:30.405880Z","iopub.execute_input":"2022-05-08T16:34:30.406149Z","iopub.status.idle":"2022-05-08T16:34:38.890875Z","shell.execute_reply.started":"2022-05-08T16:34:30.406100Z","shell.execute_reply":"2022-05-08T16:34:38.890245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"url = '../input/reviews/Restaurant_Reviews.tsv'\ndataset = pd.read_csv(url, sep='\\t', header = None)\ndataset.head()","metadata":{"_uuid":"2473ed01569b721987f1c90090e937c2f87880eb","execution":{"iopub.status.busy":"2022-05-08T18:19:32.695337Z","iopub.execute_input":"2022-05-08T18:19:32.695610Z","iopub.status.idle":"2022-05-08T18:19:32.720941Z","shell.execute_reply.started":"2022-05-08T18:19:32.695563Z","shell.execute_reply":"2022-05-08T18:19:32.720352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop the first row as it doen't worthful\ndataset = dataset.iloc[1:, :]\ndataset.reset_index(drop = True, inplace = True)\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:19:36.050053Z","iopub.execute_input":"2022-05-08T18:19:36.050353Z","iopub.status.idle":"2022-05-08T18:19:36.064198Z","shell.execute_reply.started":"2022-05-08T18:19:36.050300Z","shell.execute_reply":"2022-05-08T18:19:36.063133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is collected from yelp, where user Reviews and Recommendations of  \nBest Restaurants, Shopping, Nightlife, Food, Entertainment etc are available","metadata":{}},{"cell_type":"code","source":"dataset.tail()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:19:43.409231Z","iopub.execute_input":"2022-05-08T18:19:43.409578Z","iopub.status.idle":"2022-05-08T18:19:43.424801Z","shell.execute_reply.started":"2022-05-08T18:19:43.409525Z","shell.execute_reply":"2022-05-08T18:19:43.423538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:19:47.779177Z","iopub.execute_input":"2022-05-08T18:19:47.779622Z","iopub.status.idle":"2022-05-08T18:19:47.785909Z","shell.execute_reply.started":"2022-05-08T18:19:47.779518Z","shell.execute_reply":"2022-05-08T18:19:47.784974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:19:50.471266Z","iopub.execute_input":"2022-05-08T18:19:50.471838Z","iopub.status.idle":"2022-05-08T18:19:50.484509Z","shell.execute_reply.started":"2022-05-08T18:19:50.471779Z","shell.execute_reply":"2022-05-08T18:19:50.483391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.rename(columns={0:'Reviews', 1:'Sentiment'}, inplace=True)\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:19:54.181443Z","iopub.execute_input":"2022-05-08T18:19:54.181728Z","iopub.status.idle":"2022-05-08T18:19:54.617456Z","shell.execute_reply.started":"2022-05-08T18:19:54.181678Z","shell.execute_reply":"2022-05-08T18:19:54.616519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Reviews'][0]","metadata":{"_uuid":"b9e7580a07c762a8aaee1e8ac6ab21197e9ca3c6","execution":{"iopub.status.busy":"2022-05-08T18:20:01.400124Z","iopub.execute_input":"2022-05-08T18:20:01.400438Z","iopub.status.idle":"2022-05-08T18:20:01.407103Z","shell.execute_reply.started":"2022-05-08T18:20:01.400386Z","shell.execute_reply":"2022-05-08T18:20:01.406061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem import PorterStemmer\nstemmer = PorterStemmer()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T17:52:52.989767Z","iopub.execute_input":"2022-05-08T17:52:52.990051Z","iopub.status.idle":"2022-05-08T17:52:52.994519Z","shell.execute_reply.started":"2022-05-08T17:52:52.989999Z","shell.execute_reply":"2022-05-08T17:52:52.993470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                               u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                               u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                               u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                               u\"\\U00002500-\\U00002BEF\"  # chinese char\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U000024C2-\\U0001F251\"\n                               u\"\\U0001f926-\\U0001f937\"\n                               u\"\\U00010000-\\U0010ffff\"\n                               u\"\\u2640-\\u2642\"\n                               u\"\\u2600-\\u2B55\"\n                               u\"\\u200d\"\n                               u\"\\u23cf\"\n                               u\"\\u23e9\"\n                               u\"\\u231a\"\n                               u\"\\ufe0f\"  # dingbats\n                               u\"\\u3030\"\n                               \"]+\", flags=re.UNICODE)\nstopwords = nltk.corpus.stopwords.words('english')\ncorpus = []\nreviews = list(dataset['Reviews'])\nfor i in reviews:\n    # clean all kinds of emojies\n    emoji_clean = emoji_pattern.sub(r'', i)\n    \n    \n    single_char_clean = re.sub(r'[^a-zA-Z0-9]+', ' ', emoji_clean)\n\n    # clean punctuations, punctuations are !\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~\n    punctuation_clean = re.sub('['+string.punctuation+']', ' ', single_char_clean) \n    \n    # split each string and remove stopwords and other unnecessary words\n    split_string = punctuation_clean.split()\n    \n    stopwords_clean = [word for word in split_string if word not in (stopwords) ]\n    \n    # don't use stemmer or lemmatizer as word2vec model contains only usual words\n#     stemmed_sentences = [stemmer.stem(word.lower()) for word in stopwords_clean]\n    \n#     lemmatized_sentences = blm.lemma(stemmed_sentences)\n    \n    # join words to make complete sentence and append into the list\n    sentences  = ' '.join(stopwords_clean)\n    corpus.append(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:20:05.748058Z","iopub.execute_input":"2022-05-08T18:20:05.748360Z","iopub.status.idle":"2022-05-08T18:20:05.826842Z","shell.execute_reply.started":"2022-05-08T18:20:05.748305Z","shell.execute_reply":"2022-05-08T18:20:05.826043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus[100]","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:20:08.572781Z","iopub.execute_input":"2022-05-08T18:20:08.573094Z","iopub.status.idle":"2022-05-08T18:20:08.579132Z","shell.execute_reply.started":"2022-05-08T18:20:08.573038Z","shell.execute_reply":"2022-05-08T18:20:08.578180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Take embeddings of the words from pre-trained Word2Vec by matching our dataset words\n","metadata":{}},{"cell_type":"code","source":"# create an empty dataframe to save vectors for each sentences\nword_vectors = pd.DataFrame() \n # iterate through each document\nfor doc in pd.Series(corpus):\n    # create another empty dataframe to save vector values for each word\n    temp = pd.DataFrame()  \n    # looping through each word of a single document and spliting through space\n    for word in doc.split(' '): \n        # if word is not present in word2vec (pre-trained model) and in stopwords, then try works\n        if word not in stopwords: \n            try:    \n                word_vec = embeddings[word] \n                temp = temp.append(pd.Series(word_vec), ignore_index = True) \n            except:\n                pass\n    # temp.mean() is necessary to turn (--, 300) into (300, ) means a single column\n    word_vectors = word_vectors.append(temp.mean(), ignore_index = True) ","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:40:30.451067Z","iopub.execute_input":"2022-05-08T18:40:30.451373Z","iopub.status.idle":"2022-05-08T18:40:39.360508Z","shell.execute_reply.started":"2022-05-08T18:40:30.451315Z","shell.execute_reply":"2022-05-08T18:40:39.359701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_vectors.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:40:42.502266Z","iopub.execute_input":"2022-05-08T18:40:42.502541Z","iopub.status.idle":"2022-05-08T18:40:42.508259Z","shell.execute_reply.started":"2022-05-08T18:40:42.502491Z","shell.execute_reply":"2022-05-08T18:40:42.507452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_vectors[0]","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:40:54.510731Z","iopub.execute_input":"2022-05-08T18:40:54.511011Z","iopub.status.idle":"2022-05-08T18:40:54.519013Z","shell.execute_reply.started":"2022-05-08T18:40:54.510961Z","shell.execute_reply":"2022-05-08T18:40:54.518296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(word_vectors)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:41:07.100014Z","iopub.execute_input":"2022-05-08T18:41:07.100321Z","iopub.status.idle":"2022-05-08T18:41:07.107526Z","shell.execute_reply.started":"2022-05-08T18:41:07.100263Z","shell.execute_reply":"2022-05-08T18:41:07.106493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split dataset ","metadata":{}},{"cell_type":"code","source":"train_x, test_x, train_y, test_y = train_test_split(word_vectors,\n                                                   dataset['Sentiment'],\n                                                   test_size = 0.2,\n                                                   random_state = 1)\ntrain_x.shape, train_y.shape, test_x.shape, test_y.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:41:15.597016Z","iopub.execute_input":"2022-05-08T18:41:15.597314Z","iopub.status.idle":"2022-05-08T18:41:15.616189Z","shell.execute_reply.started":"2022-05-08T18:41:15.597255Z","shell.execute_reply":"2022-05-08T18:41:15.615360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`docs_vectors.drop('Sentiment', axis = 1)` contains docs_vectors values of (1000, 300), and it's a **DataFrame**  \nand `docs_vectors['Sentiment']` contains only sentiments of 1000 texts and it's a **column**","metadata":{}},{"cell_type":"markdown","source":"## Apply algorithm \nI used ML as it's a simple dataset","metadata":{}},{"cell_type":"code","source":"rf = RandomForestClassifier(n_estimators = 50, criterion = 'entropy', max_depth=50)\nrf.fit(train_x, train_y)\ny_pred = model.predict(test_x)\nrf_acc = accuracy_score(test_y, y_pred)\nrf_acc","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:50:09.495267Z","iopub.execute_input":"2022-05-08T18:50:09.495562Z","iopub.status.idle":"2022-05-08T18:50:10.394039Z","shell.execute_reply.started":"2022-05-08T18:50:09.495513Z","shell.execute_reply":"2022-05-08T18:50:10.393435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For Bangla text classification, we can see [this](https://rakibul-hassan.gitbook.io/deep-learning/start-page/sent_analysis1) one (use Fasttext method, not word2vec with 'bnwiki-texts-preprocessed.txt';  \notherwise it wouldn't work for huge time)","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}