{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T06:24:49.262923Z","iopub.execute_input":"2022-08-05T06:24:49.263474Z","iopub.status.idle":"2022-08-05T06:24:49.294812Z","shell.execute_reply.started":"2022-08-05T06:24:49.263365Z","shell.execute_reply":"2022-08-05T06:24:49.293711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. The goal of this competition is to classify argumentative elements in student writing as \"effective,\" \"adequate,\" or \"ineffective.\" \n2. Create a model trained on data in order to minimize bias. \n\n\n\n","metadata":{}},{"cell_type":"code","source":"#import libraries:\nimport os\nfrom os.path import join \n\nimport pandas as pd\nimport numpy as np\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom glob import glob\nfrom tqdm import tqdm\n\n\nfrom wordcloud import WordCloud, STOPWORDS\n\n\nimport nltk\nfrom nltk.corpus import stopwords\n\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport re\nimport string\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer\nfrom sklearn.metrics import confusion_matrix, recall_score, f1_score, accuracy_score, precision_score, log_loss\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:11.306864Z","iopub.execute_input":"2022-08-05T06:25:11.307361Z","iopub.status.idle":"2022-08-05T06:25:13.043470Z","shell.execute_reply.started":"2022-08-05T06:25:11.307323Z","shell.execute_reply":"2022-08-05T06:25:13.041932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the Data\n\n1. train.zip - 4191 .txt files, with each file containing the full text of an essay response in the training set\n\n2. train.csv - a .csv file containing the annotated version of all essays in the training set.\n\n3. test.csv - folder of individual .txt files, with each file containing the full text of an essay response in the test set\n\n4. sample_submission.csv - file in the required format for making predictions - note that if you are making multiple predictions for a document, submit multiple rows","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/feedback-prize-effectiveness/'","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:14.409642Z","iopub.execute_input":"2022-08-05T06:25:14.410048Z","iopub.status.idle":"2022-08-05T06:25:14.416427Z","shell.execute_reply.started":"2022-08-05T06:25:14.410017Z","shell.execute_reply":"2022-08-05T06:25:14.414704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. train.csv\nContains the annotated discourse elements for all essays in the test set. 36765 rows and 5 columns:\n1. discourse_id - ID code for discourse element\n2. essay_id - ID code for essay response. This ID code corresponds to the name of the full-text file in the train/ folder.\n3. discourse_text - Text of discourse element.\n4. discourse_type - Class label of discourse element.\n5. discourse_effectiveness - Quality rating of discourse element, the target.\n\nThere are 4191 essays.\n\nThese essays have been labelled into 36765 discourses ( a few discourses are repeated).\n\nEach discourses is labelled in one of the 3 discourse effectiveness.\n","metadata":{}},{"cell_type":"code","source":"train_df =  pd.read_csv(path + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:16.445209Z","iopub.execute_input":"2022-08-05T06:25:16.446452Z","iopub.status.idle":"2022-08-05T06:25:16.770645Z","shell.execute_reply.started":"2022-08-05T06:25:16.446394Z","shell.execute_reply":"2022-08-05T06:25:16.769246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_df.head(5))\ndisplay(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:17.073847Z","iopub.execute_input":"2022-08-05T06:25:17.074928Z","iopub.status.idle":"2022-08-05T06:25:17.103007Z","shell.execute_reply.started":"2022-08-05T06:25:17.074870Z","shell.execute_reply":"2022-08-05T06:25:17.101603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding unique values in each columns\nfor col in train_df.columns:\n    print(col + \":\" + str(len(train_df[col].unique())))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:17.825227Z","iopub.execute_input":"2022-08-05T06:25:17.825609Z","iopub.status.idle":"2022-08-05T06:25:17.897293Z","shell.execute_reply.started":"2022-08-05T06:25:17.825580Z","shell.execute_reply":"2022-08-05T06:25:17.896074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets take a look at the number of text files : it has 4191 essays\ntrain_text_files = os.listdir(path+'/train')\nlen(train_text_files)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:18.574347Z","iopub.execute_input":"2022-08-05T06:25:18.574854Z","iopub.status.idle":"2022-08-05T06:25:18.734736Z","shell.execute_reply.started":"2022-08-05T06:25:18.574790Z","shell.execute_reply":"2022-08-05T06:25:18.733766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:19.269873Z","iopub.execute_input":"2022-08-05T06:25:19.270666Z","iopub.status.idle":"2022-08-05T06:25:19.358428Z","shell.execute_reply.started":"2022-08-05T06:25:19.270628Z","shell.execute_reply":"2022-08-05T06:25:19.357150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#distribuion of discourse_effectivess -- label:\nplt.figure(figsize=(10, 5))\nsns.countplot(x=\"discourse_effectiveness\", data=train_df, order = ['Ineffective', 'Adequate', 'Effective'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:19.946490Z","iopub.execute_input":"2022-08-05T06:25:19.946903Z","iopub.status.idle":"2022-08-05T06:25:20.200601Z","shell.execute_reply.started":"2022-08-05T06:25:19.946869Z","shell.execute_reply":"2022-08-05T06:25:20.199385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#distrubution of discourse_effectiveness in discourse_type:\nplt.figure(figsize=(15, 5))\n\nsns.countplot(x = 'discourse_effectiveness',\n            hue = 'discourse_type', order = ['Ineffective', 'Adequate', 'Effective'] ,  data = train_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:20.617784Z","iopub.execute_input":"2022-08-05T06:25:20.618237Z","iopub.status.idle":"2022-08-05T06:25:20.973644Z","shell.execute_reply.started":"2022-08-05T06:25:20.618203Z","shell.execute_reply":"2022-08-05T06:25:20.972840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The target (i.e. discourse effectiveness) is highly unbalanced\n1. There are 6462 ineffective discourses\n2. There are 20977 adequate discourses\n3. There are 9326 effective discourses\n\nThe distribution of discourse-types varies for different discourse-effectiveness:\n\n1. For the Adequate and Effective it follows a similar distribution but in proportion there are more position discourses that are adequate\n2. Few Claims are Ineffective","metadata":{}},{"cell_type":"markdown","source":"This dataset is a subset of the dataset from the Feedback Prize - Evaluating Student Writing competition.\n\nFor a more deatiles eda of the discourse-types, take a look at the following notebook: \nhttps://www.kaggle.com/code/rachanabisht/evaluatingstudentwriting-complete-text-eda#Introduction:","metadata":{}},{"cell_type":"code","source":"##add columns to 'train_df' which calculates the length of string in dicourse (as dis_len) \n#and number of words of string in dicourse (as disc_word_count)\ntrain_df['disc_len'] = train_df['discourse_text'].astype(str).apply(len)\n\ntrain_df[\"disc_word_count\"] = train_df[\"discourse_text\"].apply(lambda x: len(x.split()))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:22.413781Z","iopub.execute_input":"2022-08-05T06:25:22.415247Z","iopub.status.idle":"2022-08-05T06:25:22.551216Z","shell.execute_reply.started":"2022-08-05T06:25:22.415190Z","shell.execute_reply":"2022-08-05T06:25:22.550179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:23.109780Z","iopub.execute_input":"2022-08-05T06:25:23.110198Z","iopub.status.idle":"2022-08-05T06:25:23.124005Z","shell.execute_reply.started":"2022-08-05T06:25:23.110164Z","shell.execute_reply":"2022-08-05T06:25:23.123024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets find out the average len of discourse string in various target == 'discourse_effectiveness':\ndis_str_len = train_df.groupby('discourse_effectiveness')['disc_len'].mean().sort_values()\n#plot the graph for length of prediction string per type:\ndis_str_len.plot(kind = 'barh', figsize = (10,5))\nplt.xlabel('Discourse string length')\nplt.title(' average length of disc_string per discourse effectiveness')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:23.822006Z","iopub.execute_input":"2022-08-05T06:25:23.823063Z","iopub.status.idle":"2022-08-05T06:25:24.081235Z","shell.execute_reply.started":"2022-08-05T06:25:23.823026Z","shell.execute_reply":"2022-08-05T06:25:24.079632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now lets find out the average number of words per various target == 'discourse_effectiveness':\ndis_word_num = train_df.groupby('discourse_effectiveness')['disc_word_count'].mean().sort_values()\n#plot the graph for average number of word per discourse-effectiveness:\ndis_word_num.plot(kind = 'barh', figsize = (10,5))\nplt.xlabel('average number of words')\nplt.title('Average number of word per discourse effectiveness')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:24.529948Z","iopub.execute_input":"2022-08-05T06:25:24.530401Z","iopub.status.idle":"2022-08-05T06:25:24.753999Z","shell.execute_reply.started":"2022-08-05T06:25:24.530366Z","shell.execute_reply":"2022-08-05T06:25:24.753052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text files:","metadata":{}},{"cell_type":"code","source":"# let's load all texts:\n\ntexts = []\nfor file in train_text_files :\n    with open(f'/kaggle/input/feedback-prize-effectiveness/train/{file}') as f:\n        lines = f.readlines()\n    texts.append({'id': file[:-4], 'text': ''.join(lines)})\ntexts_df = pd.DataFrame(texts)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:25.777792Z","iopub.execute_input":"2022-08-05T06:25:25.779315Z","iopub.status.idle":"2022-08-05T06:25:40.667438Z","shell.execute_reply.started":"2022-08-05T06:25:25.779265Z","shell.execute_reply":"2022-08-05T06:25:40.666098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"texts_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:40.669844Z","iopub.execute_input":"2022-08-05T06:25:40.670251Z","iopub.status.idle":"2022-08-05T06:25:40.683464Z","shell.execute_reply.started":"2022-08-05T06:25:40.670216Z","shell.execute_reply":"2022-08-05T06:25:40.681876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's look at the first text and its annotation.\ndef print_text(text_id):\n    with open(f'/kaggle/input/feedback-prize-effectiveness/train/{text_id}.txt') as f:\n        lines = f.readlines()\n    print(''.join(lines))\n    \nprint_text('87A6EF3113C6')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:40.685192Z","iopub.execute_input":"2022-08-05T06:25:40.685853Z","iopub.status.idle":"2022-08-05T06:25:40.696695Z","shell.execute_reply.started":"2022-08-05T06:25:40.685783Z","shell.execute_reply":"2022-08-05T06:25:40.694993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most used words in different Discourse Effectiveness:","metadata":{}},{"cell_type":"code","source":"train_df['discourse_text'] = train_df['discourse_text'].str.lower()\n\n#get stopwords from nltk library\nstop_english = stopwords.words(\"english\")\nother_words_to_take_out = ['school', 'students', 'people', 'would', 'could', 'many']\nstop_english.extend(other_words_to_take_out)\n\n#put dataframe of Top-10 words in dict for all discourse types\ncounts_dict = {}\nfor dt in train_df['discourse_effectiveness'].unique():\n    df = train_df.query('discourse_effectiveness == @dt')\n    text = df.discourse_text.apply(lambda x: x.split()).tolist()\n    text = [item for elem in text for item in elem]\n    df1 = pd.Series(text).value_counts().to_frame().reset_index()\n    df1.columns = ['Word', 'Frequency']\n    df1 = df1[~df1.Word.isin(stop_english)].head(10)\n    df1 = df1.set_index(\"Word\").sort_values(by = \"Frequency\", ascending = True)\n    counts_dict[dt] = df1\n\nplt.figure(figsize=(15, 12))\nplt.subplots_adjust(hspace=0.5)\n\nkeys = list(counts_dict.keys())\n\nfor n, key in enumerate(keys):\n    ax = plt.subplot(4, 2, n + 1)\n    ax.set_title(f\"Most used words in {key}\")\n    counts_dict[keys[n]].plot(ax=ax, kind = 'barh')\n    plt.ylabel(\"\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:40.699464Z","iopub.execute_input":"2022-08-05T06:25:40.699943Z","iopub.status.idle":"2022-08-05T06:25:42.294895Z","shell.execute_reply.started":"2022-08-05T06:25:40.699894Z","shell.execute_reply":"2022-08-05T06:25:42.294041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing:\nThe first step to model training is to definr X and Y input variables for the model. \n\nFor the current models we will take 'discourse_text' as X variable, later we can test the models on essay texts.\n","metadata":{}},{"cell_type":"code","source":"#convert the target variable labels into numeric '0','1','2':\neffectiveness_map = {\"Ineffective\":0, \"Adequate\":1,\"Effective\":2}\ntrain_df[\"dis_effectiveness\"] = train_df[\"discourse_effectiveness\"].map(effectiveness_map)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:42.296556Z","iopub.execute_input":"2022-08-05T06:25:42.296917Z","iopub.status.idle":"2022-08-05T06:25:42.306401Z","shell.execute_reply.started":"2022-08-05T06:25:42.296885Z","shell.execute_reply":"2022-08-05T06:25:42.305501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:42.307780Z","iopub.execute_input":"2022-08-05T06:25:42.308356Z","iopub.status.idle":"2022-08-05T06:25:42.327352Z","shell.execute_reply.started":"2022-08-05T06:25:42.308323Z","shell.execute_reply":"2022-08-05T06:25:42.326052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# text preprocessing pipeline:\ndef lowercasing(text): \n    text = \"\".join(word.lower() for word in text)\n    return text\n\ndef punctuation_es(text):\n    punctuation_words = string.punctuation + '¿¡·' \n    text = \"\".join(word for word in text if word not in punctuation_words)\n    return text\n\ndef numbers_cleanner(text):\n    text = re.sub('\\d', '', text)\n    return text\n\ndef pipeline(text):\n    text = lowercasing(text)\n    text = numbers_cleanner(text)\n    text = punctuation_es(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:42.329205Z","iopub.execute_input":"2022-08-05T06:25:42.329822Z","iopub.status.idle":"2022-08-05T06:25:42.339485Z","shell.execute_reply.started":"2022-08-05T06:25:42.329761Z","shell.execute_reply":"2022-08-05T06:25:42.338021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract text and categories from training file\nX = train_df['discourse_text']\ny = train_df['dis_effectiveness']\n\n# Preprocess and vectorize text (X)\ntfidf = TfidfVectorizer()\n\nfor i in range(len(X)):\n    X[i] = pipeline(X[i])\n\n# Vectorizer\nX = tfidf.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:25:42.341350Z","iopub.execute_input":"2022-08-05T06:25:42.342853Z","iopub.status.idle":"2022-08-05T06:26:06.110063Z","shell.execute_reply.started":"2022-08-05T06:25:42.342779Z","shell.execute_reply":"2022-08-05T06:26:06.108626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Devide train & eval data\nX_train,X_test, Y_train,Y_test = train_test_split(X,y,test_size=0.2, random_state=25)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:53:54.837536Z","iopub.execute_input":"2022-08-05T06:53:54.838787Z","iopub.status.idle":"2022-08-05T06:53:54.880328Z","shell.execute_reply.started":"2022-08-05T06:53:54.838733Z","shell.execute_reply":"2022-08-05T06:53:54.879365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:53:58.681197Z","iopub.execute_input":"2022-08-05T06:53:58.681626Z","iopub.status.idle":"2022-08-05T06:53:58.689235Z","shell.execute_reply.started":"2022-08-05T06:53:58.681587Z","shell.execute_reply":"2022-08-05T06:53:58.687494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic regression Model:\nWe will build a logistic regression model to predict the multilabel 'discouse effectiveness'","metadata":{}},{"cell_type":"code","source":"model = LogisticRegression( multi_class='ovr')\nmodel.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:58.775610Z","iopub.execute_input":"2022-08-05T06:58:58.776381Z","iopub.status.idle":"2022-08-05T06:59:16.062017Z","shell.execute_reply.started":"2022-08-05T06:58:58.776337Z","shell.execute_reply":"2022-08-05T06:59:16.059844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eval precision model\npred_eval = model.predict(X_test)\n\nprint(\"-- Eval precision:\", precision_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval recall:\", recall_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval f1:\", f1_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval accuracy:\", accuracy_score(Y_test, pred_eval))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:59:16.065943Z","iopub.execute_input":"2022-08-05T06:59:16.066626Z","iopub.status.idle":"2022-08-05T06:59:16.168195Z","shell.execute_reply.started":"2022-08-05T06:59:16.066567Z","shell.execute_reply":"2022-08-05T06:59:16.167083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\ncm_bow = confusion_matrix(Y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:59:16.172170Z","iopub.execute_input":"2022-08-05T06:59:16.172614Z","iopub.status.idle":"2022-08-05T06:59:16.690193Z","shell.execute_reply.started":"2022-08-05T06:59:16.172576Z","shell.execute_reply":"2022-08-05T06:59:16.688966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RF model: \nRandom forest model for text classification.\n\n","metadata":{}},{"cell_type":"code","source":"# Instantiate vectorizers and classifier\nvect = CountVectorizer()\ntfidf = TfidfTransformer()\nclf = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:59:23.974385Z","iopub.execute_input":"2022-08-05T06:59:23.974875Z","iopub.status.idle":"2022-08-05T06:59:23.989682Z","shell.execute_reply.started":"2022-08-05T06:59:23.974836Z","shell.execute_reply":"2022-08-05T06:59:23.988706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclf.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:59:24.849449Z","iopub.execute_input":"2022-08-05T06:59:24.850618Z","iopub.status.idle":"2022-08-05T07:03:10.264630Z","shell.execute_reply.started":"2022-08-05T06:59:24.850577Z","shell.execute_reply":"2022-08-05T07:03:10.263252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eval precision model\npred_eval = clf.predict(X_test)\n\nprint(\"-- Eval precision:\", precision_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval recall:\", recall_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval f1:\", f1_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval accuracy:\", accuracy_score(Y_test, pred_eval))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:03:10.267215Z","iopub.execute_input":"2022-08-05T07:03:10.267669Z","iopub.status.idle":"2022-08-05T07:03:11.339565Z","shell.execute_reply.started":"2022-08-05T07:03:10.267627Z","shell.execute_reply":"2022-08-05T07:03:11.337927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\ncm_bow = confusion_matrix(Y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:03:11.341636Z","iopub.execute_input":"2022-08-05T07:03:11.342198Z","iopub.status.idle":"2022-08-05T07:03:11.613821Z","shell.execute_reply.started":"2022-08-05T07:03:11.342138Z","shell.execute_reply":"2022-08-05T07:03:11.612883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Random forest shows a better performance than logistic regression for the same set of input features.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}