{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfTransformer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.preprocessing import LabelEncoder\nlabel_encoder= LabelEncoder()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T16:27:58.401380Z","iopub.execute_input":"2022-08-11T16:27:58.401859Z","iopub.status.idle":"2022-08-11T16:27:59.425236Z","shell.execute_reply.started":"2022-08-11T16:27:58.401789Z","shell.execute_reply":"2022-08-11T16:27:59.424190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.427272Z","iopub.execute_input":"2022-08-11T16:27:59.427603Z","iopub.status.idle":"2022-08-11T16:27:59.763214Z","shell.execute_reply.started":"2022-08-11T16:27:59.427570Z","shell.execute_reply":"2022-08-11T16:27:59.762354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **I will do EDA to datatrain first**","metadata":{}},{"cell_type":"code","source":"data_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.764636Z","iopub.execute_input":"2022-08-11T16:27:59.764887Z","iopub.status.idle":"2022-08-11T16:27:59.844148Z","shell.execute_reply.started":"2022-08-11T16:27:59.764864Z","shell.execute_reply":"2022-08-11T16:27:59.843180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['discourse_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.845825Z","iopub.execute_input":"2022-08-11T16:27:59.846119Z","iopub.status.idle":"2022-08-11T16:27:59.854426Z","shell.execute_reply.started":"2022-08-11T16:27:59.846092Z","shell.execute_reply":"2022-08-11T16:27:59.853458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['discourse_effectiveness'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.855888Z","iopub.execute_input":"2022-08-11T16:27:59.856506Z","iopub.status.idle":"2022-08-11T16:27:59.870497Z","shell.execute_reply.started":"2022-08-11T16:27:59.856472Z","shell.execute_reply":"2022-08-11T16:27:59.869738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.872837Z","iopub.execute_input":"2022-08-11T16:27:59.873296Z","iopub.status.idle":"2022-08-11T16:27:59.891178Z","shell.execute_reply.started":"2022-08-11T16:27:59.873272Z","shell.execute_reply":"2022-08-11T16:27:59.890549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.892260Z","iopub.execute_input":"2022-08-11T16:27:59.892716Z","iopub.status.idle":"2022-08-11T16:27:59.903374Z","shell.execute_reply.started":"2022-08-11T16:27:59.892692Z","shell.execute_reply":"2022-08-11T16:27:59.902702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using the count vectorizer\ncount = CountVectorizer()\nword_count=count.fit_transform(data_train['discourse_text'])\nprint(word_count)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:27:59.904542Z","iopub.execute_input":"2022-08-11T16:27:59.904931Z","iopub.status.idle":"2022-08-11T16:28:01.118022Z","shell.execute_reply.started":"2022-08-11T16:27:59.904906Z","shell.execute_reply":"2022-08-11T16:28:01.116799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**I will do EDA process to the text**","metadata":{}},{"cell_type":"code","source":"import re\nimport nltk\nnltk.download('stopwords')\nfrom nltk.corpus import stopwords\nfrom nltk.stem.porter import PorterStemmer\ncorpus = []\nfor i in range(0, 36765):\n  review = re.sub('[^a-zA-Z]', ' ', data_train['discourse_text'][i])\n  review = review.lower()\n  review = review.split()\n  ps = PorterStemmer()\n  all_stopwords = stopwords.words('english')\n  #all_stopwords.remove('not')\n  review = [ps.stem(word) for word in review if not word in set(all_stopwords)]\n  review = ' '.join(review)\n  corpus.append(review)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:28:01.119498Z","iopub.execute_input":"2022-08-11T16:28:01.119759Z","iopub.status.idle":"2022-08-11T16:28:53.543145Z","shell.execute_reply.started":"2022-08-11T16:28:01.119737Z","shell.execute_reply":"2022-08-11T16:28:53.542265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"let's transform the discourse effectiveness to be numerical data","metadata":{}},{"cell_type":"code","source":"data_train['discourse_effectiveness'] = label_encoder.fit_transform(data_train['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:28:53.546448Z","iopub.execute_input":"2022-08-11T16:28:53.546725Z","iopub.status.idle":"2022-08-11T16:28:53.561439Z","shell.execute_reply.started":"2022-08-11T16:28:53.546703Z","shell.execute_reply":"2022-08-11T16:28:53.560177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ncv = CountVectorizer(max_features = 36765) #36765 is the total of rows contained text\nX = cv.fit_transform(corpus).toarray()\ny = data_train.iloc[:, [4]].values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:28:53.563761Z","iopub.execute_input":"2022-08-11T16:28:53.564077Z","iopub.status.idle":"2022-08-11T16:28:54.691578Z","shell.execute_reply.started":"2022-08-11T16:28:53.564053Z","shell.execute_reply":"2022-08-11T16:28:54.690722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**First thing first, I will train the dataset to find which ML algorithm that give the best accuracy for prediction**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:28:54.692423Z","iopub.execute_input":"2022-08-11T16:28:54.692634Z","iopub.status.idle":"2022-08-11T16:28:55.921338Z","shell.execute_reply.started":"2022-08-11T16:28:54.692612Z","shell.execute_reply":"2022-08-11T16:28:55.920209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"let's try Naive Bayes","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\nclassifier = GaussianNB()\nclassifier.fit(X_train, y_train.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:28:55.922499Z","iopub.execute_input":"2022-08-11T16:28:55.922781Z","iopub.status.idle":"2022-08-11T16:29:01.253336Z","shell.execute_reply.started":"2022-08-11T16:28:55.922751Z","shell.execute_reply":"2022-08-11T16:29:01.252282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:29:01.254340Z","iopub.execute_input":"2022-08-11T16:29:01.254548Z","iopub.status.idle":"2022-08-11T16:29:02.939931Z","shell.execute_reply.started":"2022-08-11T16:29:01.254527Z","shell.execute_reply":"2022-08-11T16:29:02.939003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.concatenate((y_pred.reshape(len(y_pred),1), y_test.reshape(len(y_test),1)),1))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:29:02.940947Z","iopub.execute_input":"2022-08-11T16:29:02.941205Z","iopub.status.idle":"2022-08-11T16:29:02.948116Z","shell.execute_reply.started":"2022-08-11T16:29:02.941183Z","shell.execute_reply":"2022-08-11T16:29:02.946815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\naccuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:29:02.949276Z","iopub.execute_input":"2022-08-11T16:29:02.949593Z","iopub.status.idle":"2022-08-11T16:29:02.967655Z","shell.execute_reply.started":"2022-08-11T16:29:02.949570Z","shell.execute_reply":"2022-08-11T16:29:02.966412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"well, Naive Bayes gave the bad accuracy (behind 0.5)","metadata":{}},{"cell_type":"markdown","source":"Let's try KNN then","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nclassifier_ = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)\nclassifier_.fit(X_train, y_train.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:29:02.969165Z","iopub.execute_input":"2022-08-11T16:29:02.969434Z","iopub.status.idle":"2022-08-11T16:29:03.028377Z","shell.execute_reply.started":"2022-08-11T16:29:02.969412Z","shell.execute_reply":"2022-08-11T16:29:03.027276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_ = classifier_.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:29:03.031164Z","iopub.execute_input":"2022-08-11T16:29:03.031453Z","iopub.status.idle":"2022-08-11T16:30:41.564785Z","shell.execute_reply.started":"2022-08-11T16:29:03.031430Z","shell.execute_reply":"2022-08-11T16:30:41.563797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_ = confusion_matrix(y_test, y_pred_)\nprint(cm_)\naccuracy_score(y_test, y_pred_)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:30:41.566084Z","iopub.execute_input":"2022-08-11T16:30:41.566410Z","iopub.status.idle":"2022-08-11T16:30:41.576835Z","shell.execute_reply.started":"2022-08-11T16:30:41.566380Z","shell.execute_reply":"2022-08-11T16:30:41.575867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"not bad for KNN, I will use KNN to predict the result of datatest","metadata":{}},{"cell_type":"markdown","source":"but, why not we try to apply Decision Tree :D","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nclassifier_DTC = DecisionTreeClassifier(criterion = 'entropy', random_state = 0)\nclassifier_DTC.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:30:41.578009Z","iopub.execute_input":"2022-08-11T16:30:41.578314Z","iopub.status.idle":"2022-08-11T16:32:18.845053Z","shell.execute_reply.started":"2022-08-11T16:30:41.578283Z","shell.execute_reply":"2022-08-11T16:32:18.843918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred__=classifier_DTC.predict(X_train)\ny_pred__,y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:18.846628Z","iopub.execute_input":"2022-08-11T16:32:18.847595Z","iopub.status.idle":"2022-08-11T16:32:19.783900Z","shell.execute_reply.started":"2022-08-11T16:32:18.847546Z","shell.execute_reply":"2022-08-11T16:32:19.783280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **Let's Predict the value of The dataset**","metadata":{}},{"cell_type":"markdown","source":"**import the testing dataset**","metadata":{}},{"cell_type":"code","source":"data_test = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\ndata_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.785158Z","iopub.execute_input":"2022-08-11T16:32:19.785374Z","iopub.status.idle":"2022-08-11T16:32:19.802776Z","shell.execute_reply.started":"2022-08-11T16:32:19.785353Z","shell.execute_reply":"2022-08-11T16:32:19.801923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test['discourse_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.803883Z","iopub.execute_input":"2022-08-11T16:32:19.804128Z","iopub.status.idle":"2022-08-11T16:32:19.810219Z","shell.execute_reply.started":"2022-08-11T16:32:19.804106Z","shell.execute_reply":"2022-08-11T16:32:19.809183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.811587Z","iopub.execute_input":"2022-08-11T16:32:19.812223Z","iopub.status.idle":"2022-08-11T16:32:19.830529Z","shell.execute_reply.started":"2022-08-11T16:32:19.812189Z","shell.execute_reply":"2022-08-11T16:32:19.829639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using the count vectorizer\ncount = CountVectorizer()\nword_count_=count.fit_transform(data_test['discourse_type'])\nprint(word_count_)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.831894Z","iopub.execute_input":"2022-08-11T16:32:19.832207Z","iopub.status.idle":"2022-08-11T16:32:19.840251Z","shell.execute_reply.started":"2022-08-11T16:32:19.832176Z","shell.execute_reply":"2022-08-11T16:32:19.839219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus = []\nfor i in range(0, 10):\n  review = re.sub('[^a-zA-Z]', ' ', data_test['discourse_text'][i])\n  review = review.lower()\n  review = review.split()\n  ps = PorterStemmer()\n  all_stopwords = stopwords.words('english')\n  #all_stopwords.remove('not')\n  review = [ps.stem(word) for word in review if not word in set(all_stopwords)]\n  review = ' '.join(review)\n  corpus.append(review)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.841749Z","iopub.execute_input":"2022-08-11T16:32:19.842314Z","iopub.status.idle":"2022-08-11T16:32:19.864574Z","shell.execute_reply.started":"2022-08-11T16:32:19.842280Z","shell.execute_reply":"2022-08-11T16:32:19.863427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(corpus)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.865780Z","iopub.execute_input":"2022-08-11T16:32:19.866038Z","iopub.status.idle":"2022-08-11T16:32:19.870020Z","shell.execute_reply.started":"2022-08-11T16:32:19.866013Z","shell.execute_reply":"2022-08-11T16:32:19.869175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder=LabelEncoder()\ndata_test['discourse_type']=label_encoder.fit_transform(data_test['discourse_type'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.873744Z","iopub.execute_input":"2022-08-11T16:32:19.874390Z","iopub.status.idle":"2022-08-11T16:32:19.881375Z","shell.execute_reply.started":"2022-08-11T16:32:19.874365Z","shell.execute_reply":"2022-08-11T16:32:19.880490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv = CountVectorizer(max_features = 10)\nX_testing = cv.fit_transform(corpus).toarray()\ny_testing=data_test.iloc[:,[3]]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:19.882691Z","iopub.execute_input":"2022-08-11T16:32:19.883713Z","iopub.status.idle":"2022-08-11T16:32:19.895135Z","shell.execute_reply.started":"2022-08-11T16:32:19.883681Z","shell.execute_reply":"2022-08-11T16:32:19.894023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**I apply KNN to predict the discourse_effectiveness**","metadata":{}},{"cell_type":"markdown","source":"0 = 'Adequate'\n1 = 'Ineffective'\n2 = 'Effective'\n","metadata":{}},{"cell_type":"code","source":"classifier_new = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)\nclassifier_new.fit(X_testing,y_testing)\ny_prediction = classifier_new.predict(X_testing)\ny_prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:32:47.351383Z","iopub.execute_input":"2022-08-11T16:32:47.351775Z","iopub.status.idle":"2022-08-11T16:32:47.364807Z","shell.execute_reply.started":"2022-08-11T16:32:47.351751Z","shell.execute_reply":"2022-08-11T16:32:47.363580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'discourse_id':data_test['discourse_id'],'discourse_effectiveness':y_prediction.ravel()})\nsubmission['discourse_effectiveness'] = submission['discourse_effectiveness']\nsubmission.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:01.943062Z","iopub.execute_input":"2022-08-11T16:33:01.943363Z","iopub.status.idle":"2022-08-11T16:33:01.953316Z","shell.execute_reply.started":"2022-08-11T16:33:01.943339Z","shell.execute_reply":"2022-08-11T16:33:01.952135Z"},"trusted":true},"execution_count":null,"outputs":[]}]}