{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T05:40:56.621910Z","iopub.execute_input":"2022-07-16T05:40:56.622335Z","iopub.status.idle":"2022-07-16T05:40:59.450272Z","shell.execute_reply.started":"2022-07-16T05:40:56.622299Z","shell.execute_reply":"2022-07-16T05:40:59.449379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport string\nimport re\nfrom sklearn.model_selection import train_test_split\nfrom imblearn.combine import SMOTETomek\nfrom imblearning.under_sampling import TomekLinks\nfrom sklearn.linear_model import LogisticRegrssion\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\n","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport string\nimport re\nfrom sklearn.model_selection import train_test_split\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.metrics import confusion_matrix,recall_score,f1_score,accuracy_score,precision_score,log_loss","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.451662Z","iopub.execute_input":"2022-07-16T05:40:59.451953Z","iopub.status.idle":"2022-07-16T05:40:59.459837Z","shell.execute_reply.started":"2022-07-16T05:40:59.451925Z","shell.execute_reply":"2022-07-16T05:40:59.458749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntest=pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsample=pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.461427Z","iopub.execute_input":"2022-07-16T05:40:59.461767Z","iopub.status.idle":"2022-07-16T05:40:59.672055Z","shell.execute_reply.started":"2022-07-16T05:40:59.461738Z","shell.execute_reply":"2022-07-16T05:40:59.670667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.676110Z","iopub.execute_input":"2022-07-16T05:40:59.676970Z","iopub.status.idle":"2022-07-16T05:40:59.702493Z","shell.execute_reply.started":"2022-07-16T05:40:59.676921Z","shell.execute_reply":"2022-07-16T05:40:59.701075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effectiveness_map= {'Ineffective':0,'Adequate':1,'Effective':2}\ntrain['discourse_effectiveness']=train['discourse_effectiveness'].map(effectiveness_map)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.703841Z","iopub.execute_input":"2022-07-16T05:40:59.705170Z","iopub.status.idle":"2022-07-16T05:40:59.719252Z","shell.execute_reply.started":"2022-07-16T05:40:59.705132Z","shell.execute_reply":"2022-07-16T05:40:59.717770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.720736Z","iopub.execute_input":"2022-07-16T05:40:59.721639Z","iopub.status.idle":"2022-07-16T05:40:59.734592Z","shell.execute_reply.started":"2022-07-16T05:40:59.721600Z","shell.execute_reply":"2022-07-16T05:40:59.733339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lowercasing(text):\n    text =''.join(word.lower()for word in text)\n    return text\ndef punctuation_es(text):\n    punctuation_words = string.punctuation +'¿¡·'\n    text = ''.join(word for word in text if word not in punctuation_words)\n    return text\ndef numbers_cleanner(text):\n    text = re.sub('\\d','',text)\n    return text\ndef pipeline(text):\n    text = lowercasing(text)\n    text = numbers_cleanner(text)\n    text = punctuation_es(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.736024Z","iopub.execute_input":"2022-07-16T05:40:59.736616Z","iopub.status.idle":"2022-07-16T05:40:59.744514Z","shell.execute_reply.started":"2022-07-16T05:40:59.736578Z","shell.execute_reply":"2022-07-16T05:40:59.743517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=train['discourse_text']\ny=train['discourse_effectiveness']\n\ntfidf = TfidfVectorizer()\n\nfor i in range(len(X)):\n    X[i]=pipeline(X[i])\nX = tfidf.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:59.746243Z","iopub.execute_input":"2022-07-16T05:40:59.747210Z","iopub.status.idle":"2022-07-16T05:41:23.713321Z","shell.execute_reply.started":"2022-07-16T05:40:59.747144Z","shell.execute_reply":"2022-07-16T05:41:23.712157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resample = SMOTETomek(tomek=TomekLinks(sampling_strategy='majority'))\nX,y = resample.fit_resample(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:41:23.714498Z","iopub.execute_input":"2022-07-16T05:41:23.714918Z","iopub.status.idle":"2022-07-16T05:45:57.155163Z","shell.execute_reply.started":"2022-07-16T05:41:23.714877Z","shell.execute_reply":"2022-07-16T05:45:57.153978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train ,X_test,y_train,y_test = train_test_split(X,y,test_size=.2,random_state=25)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:45:57.159611Z","iopub.execute_input":"2022-07-16T05:45:57.160290Z","iopub.status.idle":"2022-07-16T05:45:57.186759Z","shell.execute_reply.started":"2022-07-16T05:45:57.160238Z","shell.execute_reply":"2022-07-16T05:45:57.185635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LogisticRegression(C=1000,multi_class='ovr',max_iter=10000)\nmodel.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:45:57.188261Z","iopub.execute_input":"2022-07-16T05:45:57.188633Z","iopub.status.idle":"2022-07-16T05:51:07.697129Z","shell.execute_reply.started":"2022-07-16T05:45:57.188602Z","shell.execute_reply":"2022-07-16T05:51:07.695815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_eval = model.predict(X_test)\n\nprint(\"-- eval precision:\",precision_score(y_test,pred_eval,average='weighted'))\nprint(\"-- eval recall:\",recall_score(y_test,pred_eval,average='weighted'))\nprint(\"-- eval f1:\",f1_score(y_test,pred_eval,average='weighted'))\nprint(\"--eval accuracy:\",accuracy_score(y_test,pred_eval))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:51:07.699729Z","iopub.execute_input":"2022-07-16T05:51:07.701425Z","iopub.status.idle":"2022-07-16T05:51:07.790577Z","shell.execute_reply.started":"2022-07-16T05:51:07.701379Z","shell.execute_reply":"2022-07-16T05:51:07.789269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_bow = confusion_matrix(y_test, pred_eval)\n\nclass_label = y.unique()\ndf_cm = pd.DataFrame(cm_bow, index = class_label, columns = class_label)\n\nsns.heatmap(df_cm, annot = True, fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('PRED')\nplt.ylabel('REAL')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:51:07.796949Z","iopub.execute_input":"2022-07-16T05:51:07.797691Z","iopub.status.idle":"2022-07-16T05:51:08.000482Z","shell.execute_reply.started":"2022-07-16T05:51:07.797655Z","shell.execute_reply":"2022-07-16T05:51:07.999298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_bow = confusion_matrix(y_test,pred_eval)\nclass_label = y.unique()\ndf_cm=pd.DataFrame(cm_bow,index=class_label,columns = class_label)\nsns.heatmap(df_cm,annot=True,fmt = 'd')\nplt.title('Confusion matrix')\nplt.xlabel('pred')\nplt.ylabel('real')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:51:08.002279Z","iopub.execute_input":"2022-07-16T05:51:08.003007Z","iopub.status.idle":"2022-07-16T05:51:08.204576Z","shell.execute_reply.started":"2022-07-16T05:51:08.002959Z","shell.execute_reply":"2022-07-16T05:51:08.203428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w = test['discourse_text']\nw = tfidf.transform(w)\npred_test = model.predict_proba(w)\npred_test","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:51:08.206158Z","iopub.execute_input":"2022-07-16T05:51:08.207374Z","iopub.status.idle":"2022-07-16T05:51:08.219101Z","shell.execute_reply.started":"2022-07-16T05:51:08.207317Z","shell.execute_reply":"2022-07-16T05:51:08.217953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load submission template\nsubmission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\n# Replace template with predictions\nsubmission['Ineffective'] =pred_test[:,0]\nsubmission['Adequate'] = pred_test[:,1]\nsubmission['Effective'] = pred_test[:,2]\n\n# Save submission file\nsubmission.to_csv(\"submission.csv\", index=False)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:51:08.220976Z","iopub.execute_input":"2022-07-16T05:51:08.221595Z","iopub.status.idle":"2022-07-16T05:51:08.252487Z","shell.execute_reply.started":"2022-07-16T05:51:08.221562Z","shell.execute_reply":"2022-07-16T05:51:08.251172Z"},"trusted":true},"execution_count":null,"outputs":[]}]}