{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T09:44:15.874176Z","iopub.execute_input":"2022-07-08T09:44:15.874806Z","iopub.status.idle":"2022-07-08T09:44:15.916667Z","shell.execute_reply.started":"2022-07-08T09:44:15.874676Z","shell.execute_reply":"2022-07-08T09:44:15.915680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport nltk","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:16.008002Z","iopub.execute_input":"2022-07-08T09:44:16.009064Z","iopub.status.idle":"2022-07-08T09:44:17.219848Z","shell.execute_reply.started":"2022-07-08T09:44:16.009014Z","shell.execute_reply":"2022-07-08T09:44:17.218663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= pd.read_csv('../input/sentiment-analysis-on-movie-reviews/train.tsv.zip',  sep=\"\t\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:17.221583Z","iopub.execute_input":"2022-07-08T09:44:17.222157Z","iopub.status.idle":"2022-07-08T09:44:17.530222Z","shell.execute_reply.started":"2022-07-08T09:44:17.222125Z","shell.execute_reply":"2022-07-08T09:44:17.528937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Making the reviews in lower case","metadata":{}},{"cell_type":"code","source":"train['Phrase']= train.Phrase.apply(lambda x: x.lower())\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:17.531476Z","iopub.execute_input":"2022-07-08T09:44:17.531768Z","iopub.status.idle":"2022-07-08T09:44:17.602896Z","shell.execute_reply.started":"2022-07-08T09:44:17.531742Z","shell.execute_reply":"2022-07-08T09:44:17.601693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Tokenization","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize, sent_tokenize\ntrain['words']= [nltk.word_tokenize(i) for i in train['Phrase']]\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:17.605479Z","iopub.execute_input":"2022-07-08T09:44:17.606812Z","iopub.status.idle":"2022-07-08T09:44:32.203606Z","shell.execute_reply.started":"2022-07-08T09:44:17.606766Z","shell.execute_reply":"2022-07-08T09:44:32.202429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['sentence'] = [nltk.sent_tokenize(i) for i in train['Phrase']]\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:32.204701Z","iopub.execute_input":"2022-07-08T09:44:32.204984Z","iopub.status.idle":"2022-07-08T09:44:35.776485Z","shell.execute_reply.started":"2022-07-08T09:44:32.204958Z","shell.execute_reply":"2022-07-08T09:44:35.775421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot_words= train.words.apply(len)\ntot_sent= train.sentence.apply(len)\nprint('Maximum_Words: ', tot_words.max())\nprint('Maximum_Sentences: ', tot_sent.max())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:35.777820Z","iopub.execute_input":"2022-07-08T09:44:35.778265Z","iopub.status.idle":"2022-07-08T09:44:35.871211Z","shell.execute_reply.started":"2022-07-08T09:44:35.778202Z","shell.execute_reply":"2022-07-08T09:44:35.870303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Data Cleaning","metadata":{}},{"cell_type":"code","source":"# Stopwords\nfrom nltk.corpus import stopwords\nstopwords= set(stopwords.words('english'))\ndef rem_stopword(tok):\n    out=[]\n    for t in tok:\n        if t not in stopwords:\n            out.append(t)\n    return out\n\ntrain['word2']= train.words.apply(rem_stopword)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:35.872216Z","iopub.execute_input":"2022-07-08T09:44:35.873011Z","iopub.status.idle":"2022-07-08T09:44:36.278108Z","shell.execute_reply.started":"2022-07-08T09:44:35.872973Z","shell.execute_reply":"2022-07-08T09:44:36.277015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Lemmatization","metadata":{}},{"cell_type":"code","source":"#Lemmatization requires pos tags also.\nfrom nltk.tag import PerceptronTagger\nfrom nltk.data import find\n### POS tagger\nPICKLE = \"averaged_perceptron_tagger.pickle\"\nAP_MODEL_LOC = 'file:'+str(find('taggers/averaged_perceptron_tagger/'+PICKLE))\ntagger = PerceptronTagger(load=False)\ntagger.load(AP_MODEL_LOC)\npos_tag = tagger.tag","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:36.279874Z","iopub.execute_input":"2022-07-08T09:44:36.280376Z","iopub.status.idle":"2022-07-08T09:44:36.418725Z","shell.execute_reply.started":"2022-07-08T09:44:36.280322Z","shell.execute_reply":"2022-07-08T09:44:36.417645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nlemmatizer= WordNetLemmatizer()\ndef lemm2(tok):\n    out=[]\n    for i in tok:\n        for word, tag in pos_tag([i]):\n            \n            if tag.startswith(\"NN\"):\n                out.append(lemmatizer.lemmatize(word, pos='n'))\n            elif tag.startswith('VB'):\n                out.append(lemmatizer.lemmatize(word, pos='v'))\n            elif tag.startswith('JJ'):\n                out.append(lemmatizer.lemmatize(word, pos='a'))\n            else:\n                out.append(word)\n    return out\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:36.420485Z","iopub.execute_input":"2022-07-08T09:44:36.420822Z","iopub.status.idle":"2022-07-08T09:44:36.428032Z","shell.execute_reply.started":"2022-07-08T09:44:36.420790Z","shell.execute_reply":"2022-07-08T09:44:36.426895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['lemma_word']= [lemm2(i) for i in train['word2']]\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:44:36.432436Z","iopub.execute_input":"2022-07-08T09:44:36.432788Z","iopub.status.idle":"2022-07-08T09:45:25.919172Z","shell.execute_reply.started":"2022-07-08T09:44:36.432755Z","shell.execute_reply":"2022-07-08T09:45:25.918064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nimport re\ndef rm_spl(tok):\n    out=[]\n    for w in tok:\n        out.append(re.sub('[^A-Za-z0-9]+', ' ', w))\n    return out\ntrain['lemma_word']= train['lemma_word'].apply(rm_spl)\n#train['lemma_word']= train['lemma_word'].apply(lambda x: x if x not in string.punctuation)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:25.921204Z","iopub.execute_input":"2022-07-08T09:45:25.922292Z","iopub.status.idle":"2022-07-08T09:45:27.141661Z","shell.execute_reply.started":"2022-07-08T09:45:25.922216Z","shell.execute_reply":"2022-07-08T09:45:27.140596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rm_punct(tok):\n    out=[]\n    for w in tok:\n        if w not in string.punctuation:\n            out.append(w)\n    return out\ntrain['lemma_word']= train['lemma_word'].apply(rm_punct)\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.142773Z","iopub.execute_input":"2022-07-08T09:45:27.143065Z","iopub.status.idle":"2022-07-08T09:45:27.360950Z","shell.execute_reply.started":"2022-07-08T09:45:27.143039Z","shell.execute_reply":"2022-07-08T09:45:27.359770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['phrase_2']= train.lemma_word.apply(lambda x: \" \".join(x))\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.362303Z","iopub.execute_input":"2022-07-08T09:45:27.362627Z","iopub.status.idle":"2022-07-08T09:45:27.433517Z","shell.execute_reply.started":"2022-07-08T09:45:27.362598Z","shell.execute_reply":"2022-07-08T09:45:27.432748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Sentiment.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.434638Z","iopub.execute_input":"2022-07-08T09:45:27.435085Z","iopub.status.idle":"2022-07-08T09:45:27.447312Z","shell.execute_reply.started":"2022-07-08T09:45:27.435058Z","shell.execute_reply":"2022-07-08T09:45:27.446160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= train[train['Sentiment'] != 2]\ntrain.Sentiment.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.449073Z","iopub.execute_input":"2022-07-08T09:45:27.449531Z","iopub.status.idle":"2022-07-08T09:45:27.539676Z","shell.execute_reply.started":"2022-07-08T09:45:27.449494Z","shell.execute_reply":"2022-07-08T09:45:27.538545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sentiments(x):\n    if x <2 :\n        return 0\n    else:\n        return 1\n    \ntrain.Sentiment= train.Sentiment.apply(sentiments)\ntrain.Sentiment.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.541344Z","iopub.execute_input":"2022-07-08T09:45:27.542020Z","iopub.status.idle":"2022-07-08T09:45:27.580221Z","shell.execute_reply.started":"2022-07-08T09:45:27.541976Z","shell.execute_reply":"2022-07-08T09:45:27.579475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Splitting the data\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test= train_test_split(train.phrase_2, train.Sentiment, test_size= 0.25, shuffle=True, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.581457Z","iopub.execute_input":"2022-07-08T09:45:27.581959Z","iopub.status.idle":"2022-07-08T09:45:27.598680Z","shell.execute_reply.started":"2022-07-08T09:45:27.581930Z","shell.execute_reply":"2022-07-08T09:45:27.597802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:45:27.599957Z","iopub.execute_input":"2022-07-08T09:45:27.600537Z","iopub.status.idle":"2022-07-08T09:45:27.608907Z","shell.execute_reply.started":"2022-07-08T09:45:27.600502Z","shell.execute_reply":"2022-07-08T09:45:27.607695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* CountVectorizer","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n#corpus= list(X_train)\nvectorizer= CountVectorizer()\nX= vectorizer.fit_transform(X_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:32:35.950969Z","iopub.execute_input":"2022-07-08T04:32:35.951340Z","iopub.status.idle":"2022-07-08T04:32:36.406654Z","shell.execute_reply.started":"2022-07-08T04:32:35.951310Z","shell.execute_reply":"2022-07-08T04:32:36.405665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mat= X.toarray()\ncv_df= pd.DataFrame(data= mat, columns=vectorizer.get_feature_names())\ncv_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:32:36.408195Z","iopub.execute_input":"2022-07-08T04:32:36.408803Z","iopub.status.idle":"2022-07-08T04:32:38.050374Z","shell.execute_reply.started":"2022-07-08T04:32:36.408769Z","shell.execute_reply":"2022-07-08T04:32:38.049066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Using Naive Bayes algorithm","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB\nmnb= MultinomialNB()\nmnb.fit(mat, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:32:38.052445Z","iopub.execute_input":"2022-07-08T04:32:38.052809Z","iopub.status.idle":"2022-07-08T04:33:27.742741Z","shell.execute_reply.started":"2022-07-08T04:32:38.052778Z","shell.execute_reply":"2022-07-08T04:33:27.741385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_test_mat= vectorizer.transform(X_test)\nX_test_mat= X_test_mat.toarray()\nX_test_mat","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:33:27.744415Z","iopub.execute_input":"2022-07-08T04:33:27.745236Z","iopub.status.idle":"2022-07-08T04:33:28.434331Z","shell.execute_reply.started":"2022-07-08T04:33:27.745184Z","shell.execute_reply":"2022-07-08T04:33:28.433187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_mat.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:33:28.435977Z","iopub.execute_input":"2022-07-08T04:33:28.436331Z","iopub.status.idle":"2022-07-08T04:33:28.443775Z","shell.execute_reply.started":"2022-07-08T04:33:28.436299Z","shell.execute_reply":"2022-07-08T04:33:28.442484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= mnb.predict(X_test_mat)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:33:28.445374Z","iopub.execute_input":"2022-07-08T04:33:28.445806Z","iopub.status.idle":"2022-07-08T04:33:31.797569Z","shell.execute_reply.started":"2022-07-08T04:33:28.445772Z","shell.execute_reply":"2022-07-08T04:33:31.796153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_actual= np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:33:31.799283Z","iopub.execute_input":"2022-07-08T04:33:31.800601Z","iopub.status.idle":"2022-07-08T04:33:31.807044Z","shell.execute_reply.started":"2022-07-08T04:33:31.800528Z","shell.execute_reply":"2022-07-08T04:33:31.805810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nprint(\"Accuracy: \", metrics.accuracy_score(y_actual, pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:33:31.809181Z","iopub.execute_input":"2022-07-08T04:33:31.810145Z","iopub.status.idle":"2022-07-08T04:33:31.827873Z","shell.execute_reply.started":"2022-07-08T04:33:31.810086Z","shell.execute_reply":"2022-07-08T04:33:31.826258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Using Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrf_cv= RandomForestClassifier()\nrf_cv.fit(mat, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:51.462802Z","iopub.execute_input":"2022-07-08T05:07:51.463595Z","iopub.status.idle":"2022-07-08T06:08:43.007457Z","shell.execute_reply.started":"2022-07-08T05:07:51.463553Z","shell.execute_reply":"2022-07-08T06:08:43.003549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_mat= vectorizer.transform(X_test)\nX_test_mat= X_test_mat.toarray()\nX_test_mat","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:08:43.861304Z","iopub.execute_input":"2022-07-08T06:08:43.861771Z","iopub.status.idle":"2022-07-08T06:08:44.625483Z","shell.execute_reply.started":"2022-07-08T06:08:43.861740Z","shell.execute_reply":"2022-07-08T06:08:44.624419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= rf_cv.predict(X_test_mat)\ny_actual= np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:09:28.796909Z","iopub.execute_input":"2022-07-08T06:09:28.797374Z","iopub.status.idle":"2022-07-08T06:09:46.428133Z","shell.execute_reply.started":"2022-07-08T06:09:28.797327Z","shell.execute_reply":"2022-07-08T06:09:46.426964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nprint(\"Accuracy: \", metrics.accuracy_score(y_actual, pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:09:46.430101Z","iopub.execute_input":"2022-07-08T06:09:46.430541Z","iopub.status.idle":"2022-07-08T06:09:46.440839Z","shell.execute_reply.started":"2022-07-08T06:09:46.430510Z","shell.execute_reply":"2022-07-08T06:09:46.439681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* TF-IDF","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\ntfidf_vectorizer = TfidfVectorizer(use_idf=True)\nX= tfidf_vectorizer.fit_transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:10:07.412867Z","iopub.execute_input":"2022-07-08T06:10:07.413255Z","iopub.status.idle":"2022-07-08T06:10:07.965876Z","shell.execute_reply.started":"2022-07-08T06:10:07.413223Z","shell.execute_reply":"2022-07-08T06:10:07.964907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mat= X.toarray()\ntf_df= pd.DataFrame(data= mat, columns=tfidf_vectorizer.get_feature_names())\ntf_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:10:13.472582Z","iopub.execute_input":"2022-07-08T06:10:13.472960Z","iopub.status.idle":"2022-07-08T06:10:16.030923Z","shell.execute_reply.started":"2022-07-08T06:10:13.472930Z","shell.execute_reply":"2022-07-08T06:10:16.029592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mnb= MultinomialNB()\nmnb.fit(mat, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.377927Z","iopub.status.idle":"2022-07-08T05:07:40.378290Z","shell.execute_reply.started":"2022-07-08T05:07:40.378118Z","shell.execute_reply":"2022-07-08T05:07:40.378136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_test_mat= tfidf_vectorizer.transform(X_test)\nX_test_mat= X_test_mat.toarray()\nX_test_mat","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.379458Z","iopub.status.idle":"2022-07-08T05:07:40.379801Z","shell.execute_reply.started":"2022-07-08T05:07:40.379636Z","shell.execute_reply":"2022-07-08T05:07:40.379652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_mat.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.380867Z","iopub.status.idle":"2022-07-08T05:07:40.381238Z","shell.execute_reply.started":"2022-07-08T05:07:40.381068Z","shell.execute_reply":"2022-07-08T05:07:40.381084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= mnb.predict(X_test_mat)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.382243Z","iopub.status.idle":"2022-07-08T05:07:40.382729Z","shell.execute_reply.started":"2022-07-08T05:07:40.382530Z","shell.execute_reply":"2022-07-08T05:07:40.382549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_actual= np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.383953Z","iopub.status.idle":"2022-07-08T05:07:40.384387Z","shell.execute_reply.started":"2022-07-08T05:07:40.384186Z","shell.execute_reply":"2022-07-08T05:07:40.384203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy: \", metrics.accuracy_score(y_actual, pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:07:40.386064Z","iopub.status.idle":"2022-07-08T05:07:40.386442Z","shell.execute_reply.started":"2022-07-08T05:07:40.386243Z","shell.execute_reply":"2022-07-08T05:07:40.386258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Using Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrf_tf= RandomForestClassifier()\nrf_tf.fit(mat, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:38:29.866400Z","iopub.execute_input":"2022-07-08T06:38:29.867494Z","iopub.status.idle":"2022-07-08T06:38:56.333565Z","shell.execute_reply.started":"2022-07-08T06:38:29.867456Z","shell.execute_reply":"2022-07-08T06:38:56.332305Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_mat= tfidf_vectorizer.transform(X_test)\nX_test_mat= X_test_mat.toarray()\nX_test_mat","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:38:24.065664Z","iopub.status.idle":"2022-07-08T06:38:24.066114Z","shell.execute_reply.started":"2022-07-08T06:38:24.065910Z","shell.execute_reply":"2022-07-08T06:38:24.065929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= rf_cv.predict(X_test_mat)\ny_actual= np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:38:24.067738Z","iopub.status.idle":"2022-07-08T06:38:24.068643Z","shell.execute_reply.started":"2022-07-08T06:38:24.068433Z","shell.execute_reply":"2022-07-08T06:38:24.068461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nprint(\"Accuracy: \", metrics.accuracy_score(y_actual, pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T06:38:24.069925Z","iopub.status.idle":"2022-07-08T06:38:24.070642Z","shell.execute_reply.started":"2022-07-08T06:38:24.070436Z","shell.execute_reply":"2022-07-08T06:38:24.070464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Word2Vec","metadata":{}},{"cell_type":"code","source":"import gensim","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:46:54.270899Z","iopub.execute_input":"2022-07-08T09:46:54.272041Z","iopub.status.idle":"2022-07-08T09:46:54.591653Z","shell.execute_reply.started":"2022-07-08T09:46:54.271994Z","shell.execute_reply":"2022-07-08T09:46:54.590426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['gensim_phrase']= train['phrase_2'].apply(gensim.utils.simple_preprocess)\ntrain[['phrase_2', 'gensim_phrase']].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:08:42.946836Z","iopub.execute_input":"2022-07-08T10:08:42.947291Z","iopub.status.idle":"2022-07-08T10:08:43.642580Z","shell.execute_reply.started":"2022-07-08T10:08:42.947228Z","shell.execute_reply":"2022-07-08T10:08:43.641483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gensim_phrase= train['phrase_2'].apply(gensim.utils.simple_preprocess)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:27:27.472098Z","iopub.execute_input":"2022-07-08T10:27:27.472493Z","iopub.status.idle":"2022-07-08T10:27:28.112942Z","shell.execute_reply.started":"2022-07-08T10:27:27.472461Z","shell.execute_reply":"2022-07-08T10:27:28.111729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_w2v= gensim.models.Word2Vec(gensim_phrase, vector_size= 200, min_count=2, sg=1, negative=10, seed=34)\nmodel_w2v.train(gensim_phrase, total_examples= len(train['gensim_phrase']), epochs=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:27:59.603105Z","iopub.execute_input":"2022-07-08T10:27:59.603561Z","iopub.status.idle":"2022-07-08T10:28:44.334042Z","shell.execute_reply.started":"2022-07-08T10:27:59.603526Z","shell.execute_reply":"2022-07-08T10:28:44.333300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_w2v.wv['series']","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:28:53.076974Z","iopub.execute_input":"2022-07-08T10:28:53.077359Z","iopub.status.idle":"2022-07-08T10:28:53.085803Z","shell.execute_reply.started":"2022-07-08T10:28:53.077328Z","shell.execute_reply":"2022-07-08T10:28:53.084676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_w2v.wv.most_similar(\"phone\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:23:16.622204Z","iopub.execute_input":"2022-07-08T10:23:16.622597Z","iopub.status.idle":"2022-07-08T10:23:16.643594Z","shell.execute_reply.started":"2022-07-08T10:23:16.622567Z","shell.execute_reply":"2022-07-08T10:23:16.642443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_vector(x):\n    for w in x:","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def word_vector(tokens, size):\n    vec = np.zeros(size).reshape((1, size))\n    count = 0\n    for word in tokens:\n        try:\n            vec += model_w2v.wv[word].reshape((1, size))\n            count += 1\n        except KeyError: # handling the case where the token is not in vocabulary\n            continue\n    if count != 0:\n        vec /= count\n    return vec","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:40:11.596050Z","iopub.execute_input":"2022-07-08T10:40:11.596495Z","iopub.status.idle":"2022-07-08T10:40:11.602996Z","shell.execute_reply.started":"2022-07-08T10:40:11.596407Z","shell.execute_reply":"2022-07-08T10:40:11.601852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordvec_arrays = np.zeros((len(gensim_phrase), 200)) \nfor i in range(len(gensim_phrase)):\n    wordvec_arrays[i,:] = word_vector(gensim_phrase[i], 200)\nwordvec_df = pd.DataFrame(wordvec_arrays)\nwordvec_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:40:13.481897Z","iopub.execute_input":"2022-07-08T10:40:13.482257Z","iopub.status.idle":"2022-07-08T10:40:13.541735Z","shell.execute_reply.started":"2022-07-08T10:40:13.482212Z","shell.execute_reply":"2022-07-08T10:40:13.540356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}