{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1>Logistic Regression + Tfidf + balance</h1></center>\n<center><h3>Adrián Hernández S.</h3></center>","metadata":{}},{"cell_type":"markdown","source":"### Import libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport string\nimport re\n\nfrom sklearn.model_selection import train_test_split\nfrom nltk.corpus import stopwords\n\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.metrics import confusion_matrix, recall_score, f1_score, accuracy_score, precision_score, log_loss","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:22:56.064570Z","iopub.execute_input":"2022-07-14T08:22:56.065147Z","iopub.status.idle":"2022-07-14T08:22:58.007244Z","shell.execute_reply.started":"2022-07-14T08:22:56.065004Z","shell.execute_reply":"2022-07-14T08:22:58.006257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extract data from csv & change discourse effectiveness type.","metadata":{}},{"cell_type":"code","source":"training_file = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')\ntesting_file = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')\nsubmission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\n# Lets check out training dataset\ntraining_file.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:22:58.010132Z","iopub.execute_input":"2022-07-14T08:22:58.010465Z","iopub.status.idle":"2022-07-14T08:22:58.309922Z","shell.execute_reply.started":"2022-07-14T08:22:58.010430Z","shell.execute_reply":"2022-07-14T08:22:58.309089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effectiveness_map = {\"Ineffective\":0, \"Adequate\":1,\"Effective\":2}\ntraining_file[\"discourse_effectiveness\"] = training_file[\"discourse_effectiveness\"].map(effectiveness_map)\n\n# Quantity of each discourse effectiveness\ntraining_file[\"discourse_effectiveness\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:22:58.311435Z","iopub.execute_input":"2022-07-14T08:22:58.311781Z","iopub.status.idle":"2022-07-14T08:22:58.335426Z","shell.execute_reply.started":"2022-07-14T08:22:58.311745Z","shell.execute_reply":"2022-07-14T08:22:58.334497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making preprocessing pipeline.\n   >contractions -> lowercasing -> punctuation -> numbers -> stopwords -> lemmatizer","metadata":{}},{"cell_type":"code","source":"def lowercasing(text): \n    text = \"\".join(word.lower() for word in text)\n    return text\n\ndef punctuation(text):\n    punctuation_words = string.punctuation + '¿¡·' \n    text = \"\".join(word for word in text if word not in punctuation_words)\n    return text\n\ndef numbers_cleanner(text):\n    text = re.sub('\\d', '', text)\n    return text\n\ndef pipeline(text):\n    text = lowercasing(text)\n    text = punctuation(text)\n    text = numbers_cleanner(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:22:58.338568Z","iopub.execute_input":"2022-07-14T08:22:58.338808Z","iopub.status.idle":"2022-07-14T08:22:58.346442Z","shell.execute_reply.started":"2022-07-14T08:22:58.338785Z","shell.execute_reply":"2022-07-14T08:22:58.345233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract text and categories from training file\nX = training_file['discourse_text']\ny = training_file['discourse_effectiveness']\n\n# Preprocess and vectorize text (X)\ntfidf = TfidfVectorizer()\n\nfor i in range(len(X)):\n    X[i] = pipeline(X[i])\n\n# Vectorizer\nX = tfidf.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:22:58.348212Z","iopub.execute_input":"2022-07-14T08:22:58.348676Z","iopub.status.idle":"2022-07-14T08:23:18.679134Z","shell.execute_reply.started":"2022-07-14T08:22:58.348637Z","shell.execute_reply":"2022-07-14T08:23:18.678194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Balance dataset\nresample = SMOTETomek(tomek=TomekLinks(sampling_strategy='majority'))\nX, y = resample.fit_resample(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:23:18.680502Z","iopub.execute_input":"2022-07-14T08:23:18.680842Z","iopub.status.idle":"2022-07-14T08:27:07.103245Z","shell.execute_reply.started":"2022-07-14T08:23:18.680807Z","shell.execute_reply":"2022-07-14T08:27:07.102294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Divive train & eval data\nX_train,X_test, Y_train,Y_test = train_test_split(X,y,test_size=0.2, random_state=25)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:27:07.104740Z","iopub.execute_input":"2022-07-14T08:27:07.105094Z","iopub.status.idle":"2022-07-14T08:27:07.127667Z","shell.execute_reply.started":"2022-07-14T08:27:07.105044Z","shell.execute_reply":"2022-07-14T08:27:07.126779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training precision model\nmodel = LogisticRegression(C=1000, multi_class='ovr', max_iter=10000)\nmodel.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:27:07.129272Z","iopub.execute_input":"2022-07-14T08:27:07.129632Z","iopub.status.idle":"2022-07-14T08:31:03.584804Z","shell.execute_reply.started":"2022-07-14T08:27:07.129595Z","shell.execute_reply":"2022-07-14T08:31:03.583610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Eval precision model\npred_eval = model.predict(X_test)\npred_eval_prob = model.predict_proba(X_test)\n\nprint(\"-- Eval precision:\", precision_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval recall:\", recall_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval f1:\", f1_score(Y_test, pred_eval, average='weighted'))\nprint(\"-- Eval accuracy:\", accuracy_score(Y_test, pred_eval))\n\n# Log loss eval\nprint(\"-- Eval log_loss:\", log_loss(Y_test, pred_eval_prob))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:35:45.331009Z","iopub.execute_input":"2022-07-14T08:35:45.331376Z","iopub.status.idle":"2022-07-14T08:35:45.368840Z","shell.execute_reply.started":"2022-07-14T08:35:45.331347Z","shell.execute_reply":"2022-07-14T08:35:45.367871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\ncm_bow = confusion_matrix(Y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:36:16.617168Z","iopub.execute_input":"2022-07-14T08:36:16.617507Z","iopub.status.idle":"2022-07-14T08:36:16.840964Z","shell.execute_reply.started":"2022-07-14T08:36:16.617477Z","shell.execute_reply":"2022-07-14T08:36:16.840027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"W = testing_file['discourse_text']\n\n# Vectorizer\nW = tfidf.transform(W)\n\npred_test = model.predict_proba(W)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:33:21.827428Z","iopub.execute_input":"2022-07-14T08:33:21.827755Z","iopub.status.idle":"2022-07-14T08:33:21.836996Z","shell.execute_reply.started":"2022-07-14T08:33:21.827727Z","shell.execute_reply":"2022-07-14T08:33:21.835249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.loc[:,\"Ineffective\"] = pred_test[:,0]\nsubmission.loc[:,\"Adequate\"] = pred_test[:,1]\nsubmission.loc[:,\"Effective\"] = pred_test[:,2]\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:31:03.953240Z","iopub.execute_input":"2022-07-14T08:31:03.954004Z","iopub.status.idle":"2022-07-14T08:31:03.970053Z","shell.execute_reply.started":"2022-07-14T08:31:03.953967Z","shell.execute_reply":"2022-07-14T08:31:03.969137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:31:03.971642Z","iopub.execute_input":"2022-07-14T08:31:03.972458Z","iopub.status.idle":"2022-07-14T08:31:03.979839Z","shell.execute_reply.started":"2022-07-14T08:31:03.972419Z","shell.execute_reply":"2022-07-14T08:31:03.978887Z"},"trusted":true},"execution_count":null,"outputs":[]}]}