{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom tqdm import tqdm\nfrom sklearn.svm import SVC\nfrom keras.models import Sequential\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import MultinomialNB\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nstop_words = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:36.877775Z","iopub.execute_input":"2022-07-27T18:25:36.878381Z","iopub.status.idle":"2022-07-27T18:25:36.888395Z","shell.execute_reply.started":"2022-07-27T18:25:36.878342Z","shell.execute_reply":"2022-07-27T18:25:36.886931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\nsample = pd.read_csv('../input/feedback-prize-effectiveness/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:37.379635Z","iopub.execute_input":"2022-07-27T18:25:37.380681Z","iopub.status.idle":"2022-07-27T18:25:37.548653Z","shell.execute_reply.started":"2022-07-27T18:25:37.380637Z","shell.execute_reply":"2022-07-27T18:25:37.547628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:37.893359Z","iopub.execute_input":"2022-07-27T18:25:37.894052Z","iopub.status.idle":"2022-07-27T18:25:37.905098Z","shell.execute_reply.started":"2022-07-27T18:25:37.894015Z","shell.execute_reply":"2022-07-27T18:25:37.904183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:38.336976Z","iopub.execute_input":"2022-07-27T18:25:38.337350Z","iopub.status.idle":"2022-07-27T18:25:38.348755Z","shell.execute_reply.started":"2022-07-27T18:25:38.337319Z","shell.execute_reply":"2022-07-27T18:25:38.347711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbl_enc = preprocessing.LabelEncoder()\ntrain['discourse_type'] = lbl_enc.fit_transform(train.discourse_type.values)\ny = lbl_enc.fit_transform(train.discourse_effectiveness.values)\ntest['discourse_type'] = lbl_enc.fit_transform(test.discourse_type.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:38.771260Z","iopub.execute_input":"2022-07-27T18:25:38.771801Z","iopub.status.idle":"2022-07-27T18:25:38.792827Z","shell.execute_reply.started":"2022-07-27T18:25:38.771770Z","shell.execute_reply":"2022-07-27T18:25:38.791970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain = train.discourse_text.values\nxtest = test.discourse_text.values","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:39.286506Z","iopub.execute_input":"2022-07-27T18:25:39.287203Z","iopub.status.idle":"2022-07-27T18:25:39.292671Z","shell.execute_reply.started":"2022-07-27T18:25:39.287167Z","shell.execute_reply":"2022-07-27T18:25:39.291801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (xtrain.shape)\nprint (xtest.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:25:39.936058Z","iopub.execute_input":"2022-07-27T18:25:39.936625Z","iopub.status.idle":"2022-07-27T18:25:39.941233Z","shell.execute_reply.started":"2022-07-27T18:25:39.936594Z","shell.execute_reply":"2022-07-27T18:25:39.940379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Instead of using TF-IDF, we can also use word counts as features.","metadata":{}},{"cell_type":"code","source":"ctv = CountVectorizer(analyzer='word',token_pattern=r'\\w{1,}',\n            ngram_range=(1, 3), stop_words = 'english')\n\n# Fitting Count Vectorizer to both training and test sets (semi-supervised learning)\nctv.fit(list(xtrain) + list(xtest))\nxtrain_ctv =  ctv.transform(xtrain) \nxvalid_ctv = ctv.transform(xtest)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:43:54.245702Z","iopub.execute_input":"2022-07-27T18:43:54.246149Z","iopub.status.idle":"2022-07-27T18:44:02.804186Z","shell.execute_reply.started":"2022-07-27T18:43:54.246115Z","shell.execute_reply":"2022-07-27T18:44:02.803189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting a simple Logistic Regression on Counts\nclf = LogisticRegression(C=1.0)\nclf.fit(xtrain_ctv, y)\npredictions = clf.predict_proba(xvalid_ctv)\n\n#print (\"logloss: %0.3f \" % multiclass_logloss(yvalid, predictions))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:44:15.915791Z","iopub.execute_input":"2022-07-27T18:44:15.916221Z","iopub.status.idle":"2022-07-27T18:45:26.907138Z","shell.execute_reply.started":"2022-07-27T18:44:15.916186Z","shell.execute_reply":"2022-07-27T18:45:26.905926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:45:33.363427Z","iopub.execute_input":"2022-07-27T18:45:33.365247Z","iopub.status.idle":"2022-07-27T18:45:33.382864Z","shell.execute_reply.started":"2022-07-27T18:45:33.365194Z","shell.execute_reply":"2022-07-27T18:45:33.381188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(predictions)\ndf.rename(columns = {0 : 'Ineffective', 1 : 'Adequate', 2 : 'Effective'}, inplace = True)\ndf['discourse_id'] = test['discourse_id']\ndf = df[['discourse_id', 'Ineffective', 'Adequate', 'Effective']]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:45:33.993901Z","iopub.execute_input":"2022-07-27T18:45:33.999637Z","iopub.status.idle":"2022-07-27T18:45:34.033710Z","shell.execute_reply.started":"2022-07-27T18:45:33.999353Z","shell.execute_reply":"2022-07-27T18:45:34.031943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission.to_csv('/kaggle/working/submission.csv', index = False)\ndf.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T18:45:35.086798Z","iopub.execute_input":"2022-07-27T18:45:35.087219Z","iopub.status.idle":"2022-07-27T18:45:35.094423Z","shell.execute_reply.started":"2022-07-27T18:45:35.087187Z","shell.execute_reply":"2022-07-27T18:45:35.093442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}