{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install contractions\n!pip install Sastrawi\n!pip install swifter","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:13:36.391903Z","iopub.execute_input":"2022-07-23T12:13:36.392278Z","iopub.status.idle":"2022-07-23T12:14:11.527528Z","shell.execute_reply.started":"2022-07-23T12:13:36.392196Z","shell.execute_reply":"2022-07-23T12:14:11.526588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport emoji\nimport numpy as np\nimport contractions\nimport re\nimport string","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:11.529934Z","iopub.execute_input":"2022-07-23T12:14:11.530262Z","iopub.status.idle":"2022-07-23T12:14:11.569126Z","shell.execute_reply.started":"2022-07-23T12:14:11.530229Z","shell.execute_reply":"2022-07-23T12:14:11.568144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cleaning","metadata":{}},{"cell_type":"code","source":"def remove(tweet):\n    #remove tags\n    t1 = re.sub('RT\\s', '', tweet)\n    #remove @username\n    t2 = re.sub('\\B@\\w+', \"\", t1)\n    #remove url\\\n    t3 = re.sub(r'https?:\\/\\/.*[\\r\\n]*', '', t2)\n    #remove emoji \n    t4 = emoji.replace_emoji(t3, replace='')\n    #remove word from hashtags\n    t5 = re.sub('#+', '', t4)\n    #conver into lowercase\n    t6 = t5.lower()\n    #remove looping word likes banyaaakkk\n    t7 = re.sub(r'(.)\\1+', r'\\1\\1', t6)\n    #remove punctuation\n    t8 = re.sub(r'[\\?\\/\\!]+(?=[\\?.\\!])', '', t7)\n    #remove number and special characters\n    t9 = re.sub(r'[^a-zA-Z]', ' ', t8)\n    # sentence reconstruction\n    t10 = contractions.fix(t9)\n    return t9","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:11.570274Z","iopub.execute_input":"2022-07-23T12:14:11.570908Z","iopub.status.idle":"2022-07-23T12:14:11.579040Z","shell.execute_reply.started":"2022-07-23T12:14:11.570879Z","shell.execute_reply":"2022-07-23T12:14:11.578301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/gofood-review-data-from-twitter/gofood.csv', sep=',', encoding='utf-8')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:11.581716Z","iopub.execute_input":"2022-07-23T12:14:11.582465Z","iopub.status.idle":"2022-07-23T12:14:11.777925Z","shell.execute_reply.started":"2022-07-23T12:14:11.582423Z","shell.execute_reply":"2022-07-23T12:14:11.776828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, r in df.iterrows():\n    y = remove(r['tweet'])\n    df.loc[i, 'tweet'] = y\n    \ndf.head(50)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-23T12:14:11.779345Z","iopub.execute_input":"2022-07-23T12:14:11.779692Z","iopub.status.idle":"2022-07-23T12:14:42.477652Z","shell.execute_reply.started":"2022-07-23T12:14:11.779664Z","shell.execute_reply":"2022-07-23T12:14:42.476652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tokenize","metadata":{}},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize\n\ndef word(tweet):\n    return word_tokenize(tweet)\n\ndf['token'] = df['tweet'].apply(word)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:42.478948Z","iopub.execute_input":"2022-07-23T12:14:42.480101Z","iopub.status.idle":"2022-07-23T12:14:46.709170Z","shell.execute_reply.started":"2022-07-23T12:14:42.480059Z","shell.execute_reply":"2022-07-23T12:14:46.708121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:46.710483Z","iopub.execute_input":"2022-07-23T12:14:46.710805Z","iopub.status.idle":"2022-07-23T12:14:46.724614Z","shell.execute_reply.started":"2022-07-23T12:14:46.710776Z","shell.execute_reply":"2022-07-23T12:14:46.723523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stopword removal","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords\n\nlist_stopwords = stopwords.words('indonesian')\nlist_stopwords.extend(['amp', 'nb', 'k', 'n', '   ', 'me', 'ih', 'pengin',\n                      'di', 'ke', 'wkndjanaksn', 'g', 'ga', 'gk', 'usa', 'h', \n                       'jd', 'tp','yg', 'gw', 'aj', 's', 'rb', 'rp', 't','mv', 'link', 'nb',\n                      'wtb', 'wts', 'heheheh', 'pls', 'plss', 'x', 'sj', 'ok', 'd'])\nlist_stopwords = set(list_stopwords)\n\ndef stopword_removal(words):\n    return [word for word in words if word not in list_stopwords]\n\ndf['re_stop'] = df['token'].apply(stopword_removal)\nprint(df['re_stop'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:46.725887Z","iopub.execute_input":"2022-07-23T12:14:46.726829Z","iopub.status.idle":"2022-07-23T12:14:46.929783Z","shell.execute_reply.started":"2022-07-23T12:14:46.726796Z","shell.execute_reply":"2022-07-23T12:14:46.928740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stemming","metadata":{}},{"cell_type":"code","source":"from Sastrawi.Stemmer.StemmerFactory import StemmerFactory\nimport swifter","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:46.931353Z","iopub.execute_input":"2022-07-23T12:14:46.931741Z","iopub.status.idle":"2022-07-23T12:14:48.102184Z","shell.execute_reply.started":"2022-07-23T12:14:46.931713Z","shell.execute_reply":"2022-07-23T12:14:48.101102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"factory = StemmerFactory()\nstemmer = factory.create_stemmer()\n\ndef stemmed(term):\n    return stemmer.stem(term)\n\nterm_dict = {}\n\nfor document in df['re_stop']:\n    for term in document:\n        if term not in term_dict:\n            term_dict[term] = ' '\n            \nprint(len(term_dict))\nprint(\"------------------------\")\n\nfor term in term_dict:\n    term_dict[term] = stemmed(term)\n    print(term,\":\" ,term_dict[term])\n    \nprint(term_dict)\nprint(\"------------------------\")\n\n\n# apply stemmed term to dataframe\ndef get_stemmed_term(document):\n    return [term_dict[term] for term in document]\n\ndf['stem'] = df['re_stop'].swifter.apply(get_stemmed_term)\nprint(df['stem'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:14:48.107040Z","iopub.execute_input":"2022-07-23T12:14:48.107339Z","iopub.status.idle":"2022-07-23T12:48:30.424229Z","shell.execute_reply.started":"2022-07-23T12:14:48.107312Z","shell.execute_reply":"2022-07-23T12:48:30.423514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#join sentence\n\nfor i in range(len(df['stem'])):\n    df.stem[i] = ' '.join(df.stem[i])\n    \ndf['join'] = df.stem\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:30.425171Z","iopub.execute_input":"2022-07-23T12:48:30.425837Z","iopub.status.idle":"2022-07-23T12:48:43.337217Z","shell.execute_reply.started":"2022-07-23T12:48:30.425808Z","shell.execute_reply":"2022-07-23T12:48:43.336080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from textblob import TextBlob\n\ndef getPolarity(tweet):\n    return TextBlob(tweet).sentiment.polarity\n\ndef analyze(score):\n    if score < 0:\n        return '0'\n    elif score == 0:\n        return '1'\n    else:\n        return '2'","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:43.338618Z","iopub.execute_input":"2022-07-23T12:48:43.338899Z","iopub.status.idle":"2022-07-23T12:48:43.385549Z","shell.execute_reply.started":"2022-07-23T12:48:43.338873Z","shell.execute_reply":"2022-07-23T12:48:43.384466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data = pd.DataFrame(df[['datetime', 'username', 'join']])\nfinal_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:43.389070Z","iopub.execute_input":"2022-07-23T12:48:43.389358Z","iopub.status.idle":"2022-07-23T12:48:43.410978Z","shell.execute_reply.started":"2022-07-23T12:48:43.389333Z","shell.execute_reply":"2022-07-23T12:48:43.409857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data['polarity'] = final_data['join'].apply(getPolarity)\nfinal_data['score'] = final_data['polarity'].apply(analyze)\nfinal_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:43.412091Z","iopub.execute_input":"2022-07-23T12:48:43.412362Z","iopub.status.idle":"2022-07-23T12:48:47.760960Z","shell.execute_reply.started":"2022-07-23T12:48:43.412339Z","shell.execute_reply":"2022-07-23T12:48:47.760021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data.score.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:47.762218Z","iopub.execute_input":"2022-07-23T12:48:47.762497Z","iopub.status.idle":"2022-07-23T12:48:47.772830Z","shell.execute_reply.started":"2022-07-23T12:48:47.762472Z","shell.execute_reply":"2022-07-23T12:48:47.771542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:47.775065Z","iopub.execute_input":"2022-07-23T12:48:47.776248Z","iopub.status.idle":"2022-07-23T12:48:47.782658Z","shell.execute_reply.started":"2022-07-23T12:48:47.776214Z","shell.execute_reply":"2022-07-23T12:48:47.781617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(max_features=10000)\nvectors = vectorizer.fit_transform(final_data['join'])\nwords_df = pd.DataFrame(vectors.toarray(), columns=vectorizer.get_feature_names())\nwords_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:47.784483Z","iopub.execute_input":"2022-07-23T12:48:47.785227Z","iopub.status.idle":"2022-07-23T12:48:49.110697Z","shell.execute_reply.started":"2022-07-23T12:48:47.785185Z","shell.execute_reply":"2022-07-23T12:48:49.109452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = words_df\ny = final_data.score","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:49.111737Z","iopub.execute_input":"2022-07-23T12:48:49.112034Z","iopub.status.idle":"2022-07-23T12:48:49.117256Z","shell.execute_reply.started":"2022-07-23T12:48:49.112000Z","shell.execute_reply":"2022-07-23T12:48:49.116254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import MultinomialNB","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:49.118907Z","iopub.execute_input":"2022-07-23T12:48:49.119368Z","iopub.status.idle":"2022-07-23T12:48:49.256509Z","shell.execute_reply.started":"2022-07-23T12:48:49.119339Z","shell.execute_reply":"2022-07-23T12:48:49.255324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create and train a logistic regression\nlogreg = LogisticRegression(C=100, solver='lbfgs', max_iter=1000)\nlogreg.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:49.257830Z","iopub.execute_input":"2022-07-23T12:48:49.258128Z","iopub.status.idle":"2022-07-23T12:51:58.865421Z","shell.execute_reply.started":"2022-07-23T12:48:49.258100Z","shell.execute_reply":"2022-07-23T12:51:58.864311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create and train a random forest classifier\nforest = RandomForestClassifier(n_estimators=50)\nforest.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:51:58.867497Z","iopub.execute_input":"2022-07-23T12:51:58.869436Z","iopub.status.idle":"2022-07-23T12:55:46.340803Z","shell.execute_reply.started":"2022-07-23T12:51:58.869389Z","shell.execute_reply":"2022-07-23T12:55:46.339335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create and train a linear support vector classifier (LinearSVC)\nsvc = LinearSVC()\nsvc.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:55:46.343019Z","iopub.execute_input":"2022-07-23T12:55:46.343512Z","iopub.status.idle":"2022-07-23T12:55:47.792068Z","shell.execute_reply.started":"2022-07-23T12:55:46.343460Z","shell.execute_reply":"2022-07-23T12:55:47.791228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create and train a multinomial naive bayes classifier (MultinomialNB)\nbayes = MultinomialNB()\nbayes.fit(X, y)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-23T12:55:47.793070Z","iopub.execute_input":"2022-07-23T12:55:47.793844Z","iopub.status.idle":"2022-07-23T12:55:48.475999Z","shell.execute_reply.started":"2022-07-23T12:55:47.793809Z","shell.execute_reply":"2022-07-23T12:55:48.474825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put it through the vectoriser\n\n# transform, not fit_transform, because we already learned all our words\nunknown_vectors = vectorizer.transform(final_data['join'])\nunknown_words_df = pd.DataFrame(unknown_vectors.toarray(), columns=vectorizer.get_feature_names())\nunknown_words_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:55:48.481053Z","iopub.execute_input":"2022-07-23T12:55:48.482136Z","iopub.status.idle":"2022-07-23T12:55:50.660620Z","shell.execute_reply.started":"2022-07-23T12:55:48.482086Z","shell.execute_reply":"2022-07-23T12:55:50.659391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using all our models. \n\n# Logistic Regression predictions + probabilities\nfinal_data['pred_logreg'] = logreg.predict(unknown_words_df)\nfinal_data['pred_logreg_proba'] = logreg.predict_proba(unknown_words_df)[:,1]\n\n# Random forest predictions + probabilities\nfinal_data['pred_forest'] = forest.predict(unknown_words_df)\nfinal_data['pred_forest_proba'] = forest.predict_proba(unknown_words_df)[:,1]\n\n# SVC predictions\nfinal_data['pred_svc'] = svc.predict(unknown_words_df)\n\n# Bayes predictions + probabilities\nfinal_data['pred_bayes'] = bayes.predict(unknown_words_df)\nfinal_data['pred_bayes_proba'] = bayes.predict_proba(unknown_words_df)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:55:50.662219Z","iopub.execute_input":"2022-07-23T12:55:50.663859Z","iopub.status.idle":"2022-07-23T12:56:01.292744Z","shell.execute_reply.started":"2022-07-23T12:55:50.663816Z","shell.execute_reply":"2022-07-23T12:56:01.291471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:56:01.294954Z","iopub.execute_input":"2022-07-23T12:56:01.295862Z","iopub.status.idle":"2022-07-23T12:56:01.331467Z","shell.execute_reply.started":"2022-07-23T12:56:01.295814Z","shell.execute_reply":"2022-07-23T12:56:01.330336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.8, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:56:01.333458Z","iopub.execute_input":"2022-07-23T12:56:01.334305Z","iopub.status.idle":"2022-07-23T12:56:02.020067Z","shell.execute_reply.started":"2022-07-23T12:56:01.334259Z","shell.execute_reply":"2022-07-23T12:56:02.018799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nprint(\"Training logistic regression\")\nlogreg.fit(X_train, y_train)\n\nprint(\"Training random forest\")\nforest.fit(X_train, y_train)\n\nprint(\"Training Support Vector Classifier\")\nsvc.fit(X_train, y_train)\n\nprint(\"Training Naive Bayes\")\nbayes.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:56:02.025223Z","iopub.execute_input":"2022-07-23T12:56:02.025582Z","iopub.status.idle":"2022-07-23T13:01:26.336030Z","shell.execute_reply.started":"2022-07-23T12:56:02.025539Z","shell.execute_reply":"2022-07-23T13:01:26.334840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:26.337632Z","iopub.execute_input":"2022-07-23T13:01:26.338284Z","iopub.status.idle":"2022-07-23T13:01:26.344933Z","shell.execute_reply.started":"2022-07-23T13:01:26.338238Z","shell.execute_reply":"2022-07-23T13:01:26.343701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nlogr_pred = logreg.predict(X_test)\nmatrix = confusion_matrix(y_true, logr_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:26.346692Z","iopub.execute_input":"2022-07-23T13:01:26.347492Z","iopub.status.idle":"2022-07-23T13:01:26.607443Z","shell.execute_reply.started":"2022-07-23T13:01:26.347397Z","shell.execute_reply":"2022-07-23T13:01:26.606328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nrf_pred = forest.predict(X_test)\nmatrix = confusion_matrix(y_true, rf_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:26.609714Z","iopub.execute_input":"2022-07-23T13:01:26.610445Z","iopub.status.idle":"2022-07-23T13:01:27.456986Z","shell.execute_reply.started":"2022-07-23T13:01:26.610399Z","shell.execute_reply":"2022-07-23T13:01:27.455922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nsvc_pred = svc.predict(X_test)\nmatrix = confusion_matrix(y_true, svc_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:27.458448Z","iopub.execute_input":"2022-07-23T13:01:27.459151Z","iopub.status.idle":"2022-07-23T13:01:27.665265Z","shell.execute_reply.started":"2022-07-23T13:01:27.459110Z","shell.execute_reply":"2022-07-23T13:01:27.664001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nbays_pred = bayes.predict(X_test)\nmatrix = confusion_matrix(y_true, bays_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:27.667349Z","iopub.execute_input":"2022-07-23T13:01:27.668190Z","iopub.status.idle":"2022-07-23T13:01:27.895335Z","shell.execute_reply.started":"2022-07-23T13:01:27.668144Z","shell.execute_reply":"2022-07-23T13:01:27.894215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PERCENTAGE","metadata":{}},{"cell_type":"code","source":"y_true = y_test\nlog_pred = logreg.predict(X_test)\nmatrix = confusion_matrix(y_true, log_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names).div(matrix.sum(axis=1), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:27.897307Z","iopub.execute_input":"2022-07-23T13:01:27.898152Z","iopub.status.idle":"2022-07-23T13:01:28.142441Z","shell.execute_reply.started":"2022-07-23T13:01:27.898107Z","shell.execute_reply":"2022-07-23T13:01:28.141291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nlogr_pred = logreg.predict(X_test)\nmatrix = confusion_matrix(y_true, logr_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names).div(matrix.sum(axis=1), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:28.144585Z","iopub.execute_input":"2022-07-23T13:01:28.145408Z","iopub.status.idle":"2022-07-23T13:01:28.382032Z","shell.execute_reply.started":"2022-07-23T13:01:28.145362Z","shell.execute_reply":"2022-07-23T13:01:28.380918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nrf_pred = forest.predict(X_test)\nmatrix = confusion_matrix(y_true, rf_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names).div(matrix.sum(axis=1), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:28.384021Z","iopub.execute_input":"2022-07-23T13:01:28.384832Z","iopub.status.idle":"2022-07-23T13:01:29.244901Z","shell.execute_reply.started":"2022-07-23T13:01:28.384788Z","shell.execute_reply":"2022-07-23T13:01:29.243912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nsvc_pred = svc.predict(X_test)\nmatrix = confusion_matrix(y_true, svc_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names).div(matrix.sum(axis=1), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:29.246301Z","iopub.execute_input":"2022-07-23T13:01:29.246899Z","iopub.status.idle":"2022-07-23T13:01:29.458957Z","shell.execute_reply.started":"2022-07-23T13:01:29.246858Z","shell.execute_reply":"2022-07-23T13:01:29.457824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\nbayes_pred = bayes.predict(X_test)\nmatrix = confusion_matrix(y_true, bays_pred)\n\nlabel_names = pd.Series(['negative', 'neutral', 'positive'])\npd.DataFrame(matrix,\n     columns='Predicted ' + label_names,\n     index='Is ' + label_names).div(matrix.sum(axis=1), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:29.460330Z","iopub.execute_input":"2022-07-23T13:01:29.460754Z","iopub.status.idle":"2022-07-23T13:01:29.686914Z","shell.execute_reply.started":"2022-07-23T13:01:29.460715Z","shell.execute_reply":"2022-07-23T13:01:29.685839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom yellowbrick.classifier import ClassificationReport","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:29.688217Z","iopub.execute_input":"2022-07-23T13:01:29.688863Z","iopub.status.idle":"2022-07-23T13:01:29.807818Z","shell.execute_reply.started":"2022-07-23T13:01:29.688827Z","shell.execute_reply":"2022-07-23T13:01:29.806927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, logr_pred)) #logisticregression\nprint(classification_report(y_test, rf_pred)) #randomforest\nprint(classification_report(y_test, svc_pred)) #linearsvc\nprint(classification_report(y_test, bayes_pred)) #multinomial naive bayes","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:01:29.809106Z","iopub.execute_input":"2022-07-23T13:01:29.810296Z","iopub.status.idle":"2022-07-23T13:01:30.297826Z","shell.execute_reply.started":"2022-07-23T13:01:29.810251Z","shell.execute_reply":"2022-07-23T13:01:30.296630Z"},"trusted":true},"execution_count":null,"outputs":[]}]}