{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction:\n\nThe goal of this competition is to classify argumentative elements in student writing as \"effective,\" \"adequate,\" or \"ineffective.\"\n\nCreate a model trained on data in order to minimize bias.\n\nIn my earlier notebook I have tried to apply two models Logistic rgression and Random forest to the 'discourse text' \nhttps://www.kaggle.com/code/rachanabisht/feedback-prize-lr-vs-rf\n\n\nIn the cuurent notebook I will apply the same models to the 'text' essays. In order to see how it effects the essay performance.\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T07:49:36.790536Z","iopub.execute_input":"2022-08-10T07:49:36.791371Z","iopub.status.idle":"2022-08-10T07:49:36.798426Z","shell.execute_reply.started":"2022-08-10T07:49:36.791314Z","shell.execute_reply":"2022-08-10T07:49:36.797288Z"}}},{"cell_type":"code","source":"#import libraries:\nimport os\nfrom os.path import join \n\nimport pandas as pd\nimport numpy as np\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom glob import glob\nfrom tqdm import tqdm\n\n\nfrom wordcloud import WordCloud, STOPWORDS\n\n\nimport nltk\nfrom nltk.corpus import stopwords\n\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport re\nimport string\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer\nfrom sklearn.metrics import confusion_matrix, recall_score, f1_score, accuracy_score, precision_score, log_loss","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:46:43.573788Z","iopub.execute_input":"2022-08-10T10:46:43.575275Z","iopub.status.idle":"2022-08-10T10:46:45.953132Z","shell.execute_reply.started":"2022-08-10T10:46:43.575165Z","shell.execute_reply":"2022-08-10T10:46:45.952179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/feedback-prize-effectiveness/'","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:46:45.954939Z","iopub.execute_input":"2022-08-10T10:46:45.955407Z","iopub.status.idle":"2022-08-10T10:46:45.960738Z","shell.execute_reply.started":"2022-08-10T10:46:45.955361Z","shell.execute_reply":"2022-08-10T10:46:45.959826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get dataset:\ntrain_df =  pd.read_csv(path + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:46:45.962280Z","iopub.execute_input":"2022-08-10T10:46:45.963158Z","iopub.status.idle":"2022-08-10T10:46:46.295498Z","shell.execute_reply.started":"2022-08-10T10:46:45.963105Z","shell.execute_reply":"2022-08-10T10:46:46.294476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_df.head(5))\ndisplay(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:46:46.297651Z","iopub.execute_input":"2022-08-10T10:46:46.298336Z","iopub.status.idle":"2022-08-10T10:46:46.322231Z","shell.execute_reply.started":"2022-08-10T10:46:46.298301Z","shell.execute_reply":"2022-08-10T10:46:46.320970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract data from csv\ntest_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')\nsubmission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:46:46.323773Z","iopub.execute_input":"2022-08-10T10:46:46.324559Z","iopub.status.idle":"2022-08-10T10:46:46.341995Z","shell.execute_reply.started":"2022-08-10T10:46:46.324526Z","shell.execute_reply":"2022-08-10T10:46:46.340401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:23.801919Z","iopub.execute_input":"2022-08-10T10:47:23.802370Z","iopub.status.idle":"2022-08-10T10:47:23.815199Z","shell.execute_reply.started":"2022-08-10T10:47:23.802338Z","shell.execute_reply":"2022-08-10T10:47:23.814285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:24.636817Z","iopub.execute_input":"2022-08-10T10:47:24.637491Z","iopub.status.idle":"2022-08-10T10:47:24.652514Z","shell.execute_reply.started":"2022-08-10T10:47:24.637444Z","shell.execute_reply":"2022-08-10T10:47:24.651319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets take a look at the number of text files :\ntrain_text_files = os.listdir(path+'/train')\nlen(train_text_files)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:25.412496Z","iopub.execute_input":"2022-08-10T10:47:25.412898Z","iopub.status.idle":"2022-08-10T10:47:25.554066Z","shell.execute_reply.started":"2022-08-10T10:47:25.412867Z","shell.execute_reply":"2022-08-10T10:47:25.553216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's load all texts files:\n\ntexts = []\nfor file in train_text_files :\n    with open(f'/kaggle/input/feedback-prize-effectiveness/train/{file}') as f:\n        lines = f.readlines()\n    texts.append({'id': file[:-4], 'text': ''.join(lines)})\ntexts_df = pd.DataFrame(texts)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:26.044942Z","iopub.execute_input":"2022-08-10T10:47:26.046010Z","iopub.status.idle":"2022-08-10T10:47:39.738529Z","shell.execute_reply.started":"2022-08-10T10:47:26.045971Z","shell.execute_reply":"2022-08-10T10:47:39.737420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"texts_df.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:39.740759Z","iopub.execute_input":"2022-08-10T10:47:39.741220Z","iopub.status.idle":"2022-08-10T10:47:39.749142Z","shell.execute_reply.started":"2022-08-10T10:47:39.741177Z","shell.execute_reply":"2022-08-10T10:47:39.747867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"texts_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:39.751043Z","iopub.execute_input":"2022-08-10T10:47:39.751820Z","iopub.status.idle":"2022-08-10T10:47:39.770263Z","shell.execute_reply.started":"2022-08-10T10:47:39.751732Z","shell.execute_reply":"2022-08-10T10:47:39.769067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing:\n\nThe first step to model training is to definr X and Y input variables for the model.\n\nFor the current models we will take 'text' as X variable.","metadata":{}},{"cell_type":"code","source":"#convert the target variable labels into numeric '0','1','2':\neffectiveness_map = {\"Ineffective\":0, \"Adequate\":1,\"Effective\":2}\ntrain_df[\"dis_effectiveness\"] = train_df[\"discourse_effectiveness\"].map(effectiveness_map)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:39.774457Z","iopub.execute_input":"2022-08-10T10:47:39.774960Z","iopub.status.idle":"2022-08-10T10:47:39.795458Z","shell.execute_reply.started":"2022-08-10T10:47:39.774915Z","shell.execute_reply":"2022-08-10T10:47:39.793784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# text preprocessing pipeline:\ndef lowercasing(text): \n    text = \"\".join(word.lower() for word in text)\n    return text\n\ndef punctuation_es(text):\n    punctuation_words = string.punctuation + '¿¡·' \n    text = \"\".join(word for word in text if word not in punctuation_words)\n    return text\n\ndef numbers_cleanner(text):\n    text = re.sub('\\d', '', text)\n    return text\n\ndef pipeline(text):\n    text = lowercasing(text)\n    text = numbers_cleanner(text)\n    text = punctuation_es(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:47:39.797520Z","iopub.execute_input":"2022-08-10T10:47:39.798370Z","iopub.status.idle":"2022-08-10T10:47:39.807832Z","shell.execute_reply.started":"2022-08-10T10:47:39.798321Z","shell.execute_reply":"2022-08-10T10:47:39.806477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we will rename the 'essay_id' column to 'id':\ntrain_df.rename(columns = {'essay_id':'id'}, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:40:47.752936Z","iopub.execute_input":"2022-08-10T11:40:47.753487Z","iopub.status.idle":"2022-08-10T11:40:47.759007Z","shell.execute_reply.started":"2022-08-10T11:40:47.753451Z","shell.execute_reply":"2022-08-10T11:40:47.758170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:40:48.932677Z","iopub.execute_input":"2022-08-10T11:40:48.933375Z","iopub.status.idle":"2022-08-10T11:40:48.998134Z","shell.execute_reply.started":"2022-08-10T11:40:48.933326Z","shell.execute_reply":"2022-08-10T11:40:48.996915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Merge train_df to texts_df on 'id':\nresult_df = pd.merge(train_df, texts_df, on=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:39:16.292767Z","iopub.execute_input":"2022-08-10T11:39:16.293586Z","iopub.status.idle":"2022-08-10T11:39:16.321096Z","shell.execute_reply.started":"2022-08-10T11:39:16.293553Z","shell.execute_reply":"2022-08-10T11:39:16.320265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:39:23.829221Z","iopub.execute_input":"2022-08-10T11:39:23.829614Z","iopub.status.idle":"2022-08-10T11:39:23.836700Z","shell.execute_reply.started":"2022-08-10T11:39:23.829584Z","shell.execute_reply":"2022-08-10T11:39:23.835672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:39:34.077319Z","iopub.execute_input":"2022-08-10T11:39:34.077722Z","iopub.status.idle":"2022-08-10T11:39:34.092601Z","shell.execute_reply.started":"2022-08-10T11:39:34.077690Z","shell.execute_reply":"2022-08-10T11:39:34.091674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract text and categories from training file\nX = result_df['text']\ny =result_df['dis_effectiveness']\n\n# Preprocess and vectorize text (X)\ntfidf = TfidfVectorizer()\n\nfor i in range(len(X)):\n    X[i] = pipeline(X[i])\n\n# Vectorizer\nX = tfidf.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:43:36.024231Z","iopub.execute_input":"2022-08-10T11:43:36.024664Z","iopub.status.idle":"2022-08-10T11:44:37.701017Z","shell.execute_reply.started":"2022-08-10T11:43:36.024628Z","shell.execute_reply":"2022-08-10T11:44:37.699535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Devide train & eval data\nX_train,X_test, Y_train,Y_test = train_test_split(X,y,test_size=0.2, random_state=25)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:44:43.703162Z","iopub.execute_input":"2022-08-10T11:44:43.703716Z","iopub.status.idle":"2022-08-10T11:44:43.747336Z","shell.execute_reply.started":"2022-08-10T11:44:43.703671Z","shell.execute_reply":"2022-08-10T11:44:43.745849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:44:45.675223Z","iopub.execute_input":"2022-08-10T11:44:45.675631Z","iopub.status.idle":"2022-08-10T11:44:45.682170Z","shell.execute_reply.started":"2022-08-10T11:44:45.675600Z","shell.execute_reply":"2022-08-10T11:44:45.681145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic regression Model:¶\nWe will build a logistic regression model to predict the multilabel 'discouse effectiveness'","metadata":{}},{"cell_type":"code","source":"model = LogisticRegression( multi_class='ovr')\nmodel.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:44:50.203179Z","iopub.execute_input":"2022-08-10T11:44:50.204557Z","iopub.status.idle":"2022-08-10T11:45:20.026592Z","shell.execute_reply.started":"2022-08-10T11:44:50.204511Z","shell.execute_reply":"2022-08-10T11:45:20.024908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eval precision model\npred_eval = model.predict(X_test)\n\nprint(\"-- Eval precision:\", precision_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval recall:\", recall_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval f1:\", f1_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval accuracy:\", accuracy_score(Y_test, pred_eval))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:45:20.029773Z","iopub.execute_input":"2022-08-10T11:45:20.030232Z","iopub.status.idle":"2022-08-10T11:45:20.099188Z","shell.execute_reply.started":"2022-08-10T11:45:20.030187Z","shell.execute_reply":"2022-08-10T11:45:20.097572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\ncm_bow = confusion_matrix(Y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:45:20.107142Z","iopub.execute_input":"2022-08-10T11:45:20.111745Z","iopub.status.idle":"2022-08-10T11:45:20.356817Z","shell.execute_reply.started":"2022-08-10T11:45:20.111664Z","shell.execute_reply":"2022-08-10T11:45:20.355543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RF model:\nRandom forest model for text classification.","metadata":{}},{"cell_type":"code","source":"# Instantiate vectorizers and classifier\nvect = CountVectorizer()\ntfidf = TfidfTransformer()\nclf = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:45:51.335214Z","iopub.execute_input":"2022-08-10T11:45:51.336215Z","iopub.status.idle":"2022-08-10T11:45:51.343404Z","shell.execute_reply.started":"2022-08-10T11:45:51.336177Z","shell.execute_reply":"2022-08-10T11:45:51.342119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:45:55.858444Z","iopub.execute_input":"2022-08-10T11:45:55.859047Z","iopub.status.idle":"2022-08-10T12:33:57.534649Z","shell.execute_reply.started":"2022-08-10T11:45:55.859003Z","shell.execute_reply":"2022-08-10T12:33:57.533580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eval precision model\npred_eval = clf.predict(X_test)\n\nprint(\"-- Eval precision:\", precision_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval recall:\", recall_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval f1:\", f1_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval accuracy:\", accuracy_score(Y_test, pred_eval))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:44:34.368790Z","iopub.execute_input":"2022-08-10T12:44:34.370005Z","iopub.status.idle":"2022-08-10T12:44:35.173537Z","shell.execute_reply.started":"2022-08-10T12:44:34.369955Z","shell.execute_reply":"2022-08-10T12:44:35.172198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\ncm_bow = confusion_matrix(Y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:44:35.175975Z","iopub.execute_input":"2022-08-10T12:44:35.176460Z","iopub.status.idle":"2022-08-10T12:44:35.387788Z","shell.execute_reply.started":"2022-08-10T12:44:35.176416Z","shell.execute_reply":"2022-08-10T12:44:35.386428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"there is ana improvement with using'essays' versus 'discourse text as the model input.\nPlease refer to the previous notebook https://www.kaggle.com/code/rachanabisht/feedback-prize-lr-vs-rf\n\nthere is an improvement of 10% in the models!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}