{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"82bacf83-3c84-4970-b0f1-f5c0f0f44d72","_cell_guid":"64d37334-1577-4ffe-8a9e-aa3b805982b5","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:04.296590Z","iopub.execute_input":"2022-07-16T11:30:04.297087Z","iopub.status.idle":"2022-07-16T11:30:05.741458Z","shell.execute_reply.started":"2022-07-16T11:30:04.296981Z","shell.execute_reply":"2022-07-16T11:30:05.740663Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport string\nimport re\nfrom sklearn.model_selection import train_test_split\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.metrics import confusion_matrix,recall_score,f1_score,accuracy_score,precision_score,log_loss","metadata":{"_uuid":"04827bf5-7c31-4f86-b367-41a3dde8ea95","_cell_guid":"f842b7b0-562a-4770-badf-750e8f65881f","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:05.742860Z","iopub.execute_input":"2022-07-16T11:30:05.743176Z","iopub.status.idle":"2022-07-16T11:30:07.705399Z","shell.execute_reply.started":"2022-07-16T11:30:05.743148Z","shell.execute_reply":"2022-07-16T11:30:07.704097Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntest=pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsample=pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")","metadata":{"_uuid":"9616e045-14f9-43a1-ae42-8909b9c2752a","_cell_guid":"6f7ff85b-42a9-403d-88cd-5586b94835a8","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:07.707474Z","iopub.execute_input":"2022-07-16T11:30:07.707907Z","iopub.status.idle":"2022-07-16T11:30:08.048206Z","shell.execute_reply.started":"2022-07-16T11:30:07.707856Z","shell.execute_reply":"2022-07-16T11:30:08.047213Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"_uuid":"a5713488-4a2f-4fbd-b296-c04a09cb9f4d","_cell_guid":"3a3efd62-71a9-415e-bd46-2c0534dea330","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:08.050430Z","iopub.execute_input":"2022-07-16T11:30:08.051476Z","iopub.status.idle":"2022-07-16T11:30:08.074919Z","shell.execute_reply.started":"2022-07-16T11:30:08.051436Z","shell.execute_reply":"2022-07-16T11:30:08.073986Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effectiveness_map= {'Ineffective':0,'Adequate':1,'Effective':2}\ntrain['discourse_effectiveness']=train['discourse_effectiveness'].map(effectiveness_map)","metadata":{"_uuid":"b15d76b6-426c-4400-a7be-0e384fdedc89","_cell_guid":"00e0179b-01d2-4e47-9b2a-55b47777e3e1","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:08.076393Z","iopub.execute_input":"2022-07-16T11:30:08.077060Z","iopub.status.idle":"2022-07-16T11:30:08.097365Z","shell.execute_reply.started":"2022-07-16T11:30:08.077025Z","shell.execute_reply":"2022-07-16T11:30:08.096360Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"_uuid":"1aa11591-f595-4a35-90fa-d3433357ae09","_cell_guid":"9d057428-1d00-4683-afb1-4769c79c3523","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:08.098822Z","iopub.execute_input":"2022-07-16T11:30:08.099415Z","iopub.status.idle":"2022-07-16T11:30:08.111687Z","shell.execute_reply.started":"2022-07-16T11:30:08.099374Z","shell.execute_reply":"2022-07-16T11:30:08.110770Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lowercasing(text):\n    text =''.join(word.lower()for word in text)\n    return text\ndef punctuation_es(text):\n    punctuation_words = string.punctuation +'¿¡·'\n    text = ''.join(word for word in text if word not in punctuation_words)\n    return text\ndef numbers_cleanner(text):\n    text = re.sub('\\d','',text)\n    return text\ndef pipeline(text):\n    text = lowercasing(text)\n    text = numbers_cleanner(text)\n    text = punctuation_es(text)\n    return text","metadata":{"_uuid":"0bbd9f94-8094-4c57-9dc5-f46351afb9ce","_cell_guid":"ac38acb1-db69-432e-a094-95d50af4a634","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:08.112972Z","iopub.execute_input":"2022-07-16T11:30:08.113821Z","iopub.status.idle":"2022-07-16T11:30:08.124974Z","shell.execute_reply.started":"2022-07-16T11:30:08.113786Z","shell.execute_reply":"2022-07-16T11:30:08.123738Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=train['discourse_text']\ny=train['discourse_effectiveness']\n\ntfidf = TfidfVectorizer()\n\nfor i in range(len(X)):\n    X[i]=pipeline(X[i])\nX = tfidf.fit_transform(X)","metadata":{"_uuid":"b5f92997-0a96-4837-9ff0-d55e393cb9e9","_cell_guid":"3d27c6e8-d9f3-416d-a9af-8fca20a7992b","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:08.126794Z","iopub.execute_input":"2022-07-16T11:30:08.127526Z","iopub.status.idle":"2022-07-16T11:30:31.521790Z","shell.execute_reply.started":"2022-07-16T11:30:08.127477Z","shell.execute_reply":"2022-07-16T11:30:31.520432Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resample = SMOTETomek(tomek=TomekLinks(sampling_strategy='majority'))\nX,y = resample.fit_resample(X,y)","metadata":{"_uuid":"052c1eca-10d0-4776-ad79-ec31d33a267f","_cell_guid":"ef3df48c-ecbe-43e2-9bda-2d51dd6998fb","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:30:31.523315Z","iopub.execute_input":"2022-07-16T11:30:31.523728Z","iopub.status.idle":"2022-07-16T11:35:07.078728Z","shell.execute_reply.started":"2022-07-16T11:30:31.523692Z","shell.execute_reply":"2022-07-16T11:35:07.077711Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train ,X_test,y_train,y_test = train_test_split(X,y,test_size=.2,random_state=25)","metadata":{"_uuid":"7fa577ec-d89d-4f6a-b6cc-9297e5bd9326","_cell_guid":"905278c2-402f-41cd-b050-f073f828e992","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:35:07.083550Z","iopub.execute_input":"2022-07-16T11:35:07.084564Z","iopub.status.idle":"2022-07-16T11:35:07.115598Z","shell.execute_reply.started":"2022-07-16T11:35:07.084487Z","shell.execute_reply":"2022-07-16T11:35:07.114377Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LogisticRegression(C=1000,multi_class='ovr',max_iter=10000)\nmodel.fit(X_train,y_train)","metadata":{"_uuid":"ca7aa633-36c7-4f77-bce2-c34a6df35ee0","_cell_guid":"a8406614-b606-4f40-a55f-f6a47087ecca","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:35:07.116942Z","iopub.execute_input":"2022-07-16T11:35:07.117258Z","iopub.status.idle":"2022-07-16T11:39:53.469114Z","shell.execute_reply.started":"2022-07-16T11:35:07.117231Z","shell.execute_reply":"2022-07-16T11:39:53.467663Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_eval = model.predict(X_test)\n\nprint(\"-- eval precision:\",precision_score(y_test,pred_eval,average='weighted'))\nprint(\"-- eval recall:\",recall_score(y_test,pred_eval,average='weighted'))\nprint(\"-- eval f1:\",f1_score(y_test,pred_eval,average='weighted'))\nprint(\"--eval accuracy:\",accuracy_score(y_test,pred_eval))","metadata":{"_uuid":"8a7b6aae-3097-410a-9eba-349133f827e1","_cell_guid":"4ea9f306-6bef-4e51-b5bf-768aaf6f0f3a","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:39:53.471077Z","iopub.execute_input":"2022-07-16T11:39:53.473743Z","iopub.status.idle":"2022-07-16T11:39:53.570890Z","shell.execute_reply.started":"2022-07-16T11:39:53.473691Z","shell.execute_reply":"2022-07-16T11:39:53.569823Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_bow = confusion_matrix(y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"_uuid":"a2747388-fb1e-4739-a3bf-a2d25af28642","_cell_guid":"09f8afbb-c2bf-4657-90ad-90fc9ac7891c","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:39:53.572723Z","iopub.execute_input":"2022-07-16T11:39:53.573177Z","iopub.status.idle":"2022-07-16T11:39:53.888703Z","shell.execute_reply.started":"2022-07-16T11:39:53.573132Z","shell.execute_reply":"2022-07-16T11:39:53.887863Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_bow = confusion_matrix(y_test,pred_eval)\nclass_label = y.unique()\ndf_cm=pd.DataFrame(cm_bow,index=class_label,columns = class_label)\nsns.heatmap(df_cm,annot=True,fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('pred')\nplt.ylabel('real')\nplt.show()","metadata":{"_uuid":"1fcfb163-c108-42f8-b5e2-a112eaca1d6b","_cell_guid":"a0e86153-2056-4010-ade3-30b78764d41b","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:39:53.890041Z","iopub.execute_input":"2022-07-16T11:39:53.890889Z","iopub.status.idle":"2022-07-16T11:39:54.146823Z","shell.execute_reply.started":"2022-07-16T11:39:53.890851Z","shell.execute_reply":"2022-07-16T11:39:54.145962Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w = test['discourse_text']\nw = tfidf.transform(w)\npred_test = model.predict_proba(w)\npred_test","metadata":{"_uuid":"e44d266a-09da-4c08-bfa7-0cea6a14d58c","_cell_guid":"a3857c07-0582-4865-b542-9728fdbda6e3","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:39:54.148329Z","iopub.execute_input":"2022-07-16T11:39:54.148928Z","iopub.status.idle":"2022-07-16T11:39:54.163494Z","shell.execute_reply.started":"2022-07-16T11:39:54.148893Z","shell.execute_reply":"2022-07-16T11:39:54.162574Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load submission template\nsubmission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\n# Replace template with predictions\nsubmission['Ineffective'] =pred_test[:,0]\nsubmission['Adequate'] = pred_test[:,1]\nsubmission['Effective'] = pred_test[:,2]\n\n# Save submission file\nsubmission.to_csv(\"submission.csv\", index=False)\n\nsubmission.head()","metadata":{"_uuid":"7b6e4bb0-e2ca-406c-8818-ce090dd16791","_cell_guid":"9a3a2201-4fa6-4be9-9116-05ed4fa29925","collapsed":false,"execution":{"iopub.status.busy":"2022-07-16T11:39:54.164923Z","iopub.execute_input":"2022-07-16T11:39:54.165456Z","iopub.status.idle":"2022-07-16T11:39:54.194972Z","shell.execute_reply.started":"2022-07-16T11:39:54.165424Z","shell.execute_reply":"2022-07-16T11:39:54.194058Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}