{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-10T10:10:06.475187Z","iopub.execute_input":"2022-07-10T10:10:06.475577Z","iopub.status.idle":"2022-07-10T10:10:07.913212Z","shell.execute_reply.started":"2022-07-10T10:10:06.475498Z","shell.execute_reply":"2022-07-10T10:10:07.912032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom time import time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom bs4 import BeautifulSoup\nimport shutil\nimport gensim\nimport re\nimport spacy","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:07.918460Z","iopub.execute_input":"2022-07-10T10:10:07.920853Z","iopub.status.idle":"2022-07-10T10:10:11.396762Z","shell.execute_reply.started":"2022-07-10T10:10:07.920813Z","shell.execute_reply":"2022-07-10T10:10:11.395822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, GRU,SimpleRNN\nfrom tensorflow.keras.layers import Dense, Activation, Dropout\nfrom tensorflow.keras.layers import Embedding\nfrom tensorflow.keras.layers import BatchNormalization\nfrom tensorflow.python.keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom tensorflow.keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom tensorflow.keras.preprocessing import sequence, text\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:11.398454Z","iopub.execute_input":"2022-07-10T10:10:11.398853Z","iopub.status.idle":"2022-07-10T10:10:19.693807Z","shell.execute_reply.started":"2022-07-10T10:10:11.398816Z","shell.execute_reply":"2022-07-10T10:10:19.692844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:19.696335Z","iopub.execute_input":"2022-07-10T10:10:19.696673Z","iopub.status.idle":"2022-07-10T10:10:19.715106Z","shell.execute_reply.started":"2022-07-10T10:10:19.696638Z","shell.execute_reply":"2022-07-10T10:10:19.713950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntest_df = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsubmission_sample = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:19.717204Z","iopub.execute_input":"2022-07-10T10:10:19.717872Z","iopub.status.idle":"2022-07-10T10:10:19.981271Z","shell.execute_reply.started":"2022-07-10T10:10:19.717834Z","shell.execute_reply":"2022-07-10T10:10:19.980271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:19.982703Z","iopub.execute_input":"2022-07-10T10:10:19.983082Z","iopub.status.idle":"2022-07-10T10:10:20.008120Z","shell.execute_reply.started":"2022-07-10T10:10:19.983038Z","shell.execute_reply":"2022-07-10T10:10:20.007070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.009583Z","iopub.execute_input":"2022-07-10T10:10:20.010417Z","iopub.status.idle":"2022-07-10T10:10:20.053130Z","shell.execute_reply.started":"2022-07-10T10:10:20.010389Z","shell.execute_reply":"2022-07-10T10:10:20.051853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.054754Z","iopub.execute_input":"2022-07-10T10:10:20.055076Z","iopub.status.idle":"2022-07-10T10:10:20.083950Z","shell.execute_reply.started":"2022-07-10T10:10:20.055050Z","shell.execute_reply":"2022-07-10T10:10:20.082911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape\ntest_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.085366Z","iopub.execute_input":"2022-07-10T10:10:20.085924Z","iopub.status.idle":"2022-07-10T10:10:20.094847Z","shell.execute_reply.started":"2022-07-10T10:10:20.085885Z","shell.execute_reply":"2022-07-10T10:10:20.093748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"discourse_type\"].value_counts().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.100009Z","iopub.execute_input":"2022-07-10T10:10:20.100335Z","iopub.status.idle":"2022-07-10T10:10:20.312949Z","shell.execute_reply.started":"2022-07-10T10:10:20.100292Z","shell.execute_reply":"2022-07-10T10:10:20.312036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,4))\nsns.countplot(x='discourse_type', data=train_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.314353Z","iopub.execute_input":"2022-07-10T10:10:20.316346Z","iopub.status.idle":"2022-07-10T10:10:20.534540Z","shell.execute_reply.started":"2022-07-10T10:10:20.316315Z","shell.execute_reply":"2022-07-10T10:10:20.533640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,4))\nsns.countplot(x='discourse_effectiveness', data=train_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.535881Z","iopub.execute_input":"2022-07-10T10:10:20.536225Z","iopub.status.idle":"2022-07-10T10:10:20.739815Z","shell.execute_reply.started":"2022-07-10T10:10:20.536187Z","shell.execute_reply":"2022-07-10T10:10:20.738773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp = spacy.load('en_core_web_sm')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:20.741245Z","iopub.execute_input":"2022-07-10T10:10:20.742321Z","iopub.status.idle":"2022-07-10T10:10:21.779454Z","shell.execute_reply.started":"2022-07-10T10:10:20.742280Z","shell.execute_reply":"2022-07-10T10:10:21.778220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(q):\n    \n    q = str(q).lower().strip()\n    contractions = { \n    \"ain't\": \"am not\",\n    \"aren't\": \"are not\",\n    \"can't\": \"can not\",\n    \"can't've\": \"can not have\",\n    \"'cause\": \"because\",\n    \"could've\": \"could have\",\n    \"couldn't\": \"could not\",\n    \"couldn't've\": \"could not have\",\n    \"didn't\": \"did not\",\n    \"doesn't\": \"does not\",\n    \"don't\": \"do not\",\n    \"hadn't\": \"had not\",\n    \"hadn't've\": \"had not have\",\n    \"hasn't\": \"has not\",\n    \"haven't\": \"have not\",\n    \"he'd\": \"he would\",\n    \"he'd've\": \"he would have\",\n    \"he'll\": \"he will\",\n    \"he'll've\": \"he will have\",\n    \"he's\": \"he is\",\n    \"how'd\": \"how did\",\n    \"how'd'y\": \"how do you\",\n    \"how'll\": \"how will\",\n    \"how's\": \"how is\",\n    \"i'd\": \"i would\",\n    \"i'd've\": \"i would have\",\n    \"i'll\": \"i will\",\n    \"i'll've\": \"i will have\",\n    \"i'm\": \"i am\",\n    \"i've\": \"i have\",\n    \"isn't\": \"is not\",\n    \"it'd\": \"it would\",\n    \"it'd've\": \"it would have\",\n    \"it'll\": \"it will\",\n    \"it'll've\": \"it will have\",\n    \"it's\": \"it is\",\n    \"let's\": \"let us\",\n    \"ma'am\": \"madam\",\n    \"mayn't\": \"may not\",\n    \"might've\": \"might have\",\n    \"mightn't\": \"might not\",\n    \"mightn't've\": \"might not have\",\n    \"must've\": \"must have\",\n    \"mustn't\": \"must not\",\n    \"mustn't've\": \"must not have\",\n    \"needn't\": \"need not\",\n    \"needn't've\": \"need not have\",\n    \"o'clock\": \"of the clock\",\n    \"oughtn't\": \"ought not\",\n    \"oughtn't've\": \"ought not have\",\n    \"shan't\": \"shall not\",\n    \"sha'n't\": \"shall not\",\n    \"shan't've\": \"shall not have\",\n    \"she'd\": \"she would\",\n    \"she'd've\": \"she would have\",\n    \"she'll\": \"she will\",\n    \"she'll've\": \"she will have\",\n    \"she's\": \"she is\",\n    \"should've\": \"should have\",\n    \"shouldn't\": \"should not\",\n    \"shouldn't've\": \"should not have\",\n    \"so've\": \"so have\",\n    \"so's\": \"so as\",\n    \"that'd\": \"that would\",\n    \"that'd've\": \"that would have\",\n    \"that's\": \"that is\",\n    \"there'd\": \"there would\",\n    \"there'd've\": \"there would have\",\n    \"there's\": \"there is\",\n    \"they'd\": \"they would\",\n    \"they'd've\": \"they would have\",\n    \"they'll\": \"they will\",\n    \"they'll've\": \"they will have\",\n    \"they're\": \"they are\",\n    \"they've\": \"they have\",\n    \"to've\": \"to have\",\n    \"wasn't\": \"was not\",\n    \"we'd\": \"we would\",\n    \"we'd've\": \"we would have\",\n    \"we'll\": \"we will\",\n    \"we'll've\": \"we will have\",\n    \"we're\": \"we are\",\n    \"we've\": \"we have\",\n    \"weren't\": \"were not\",\n    \"what'll\": \"what will\",\n    \"what'll've\": \"what will have\",\n    \"what're\": \"what are\",\n    \"what's\": \"what is\",\n    \"what've\": \"what have\",\n    \"when's\": \"when is\",\n    \"when've\": \"when have\",\n    \"where'd\": \"where did\",\n    \"where's\": \"where is\",\n    \"where've\": \"where have\",\n    \"who'll\": \"who will\",\n    \"who'll've\": \"who will have\",\n    \"who's\": \"who is\",\n    \"who've\": \"who have\",\n    \"why's\": \"why is\",\n    \"why've\": \"why have\",\n    \"will've\": \"will have\",\n    \"won't\": \"will not\",\n    \"won't've\": \"will not have\",\n    \"would've\": \"would have\",\n    \"wouldn't\": \"would not\",\n    \"wouldn't've\": \"would not have\",\n    \"y'all\": \"you all\",\n    \"y'all'd\": \"you all would\",\n    \"y'all'd've\": \"you all would have\",\n    \"y'all're\": \"you all are\",\n    \"y'all've\": \"you all have\",\n    \"you'd\": \"you would\",\n    \"you'd've\": \"you would have\",\n    \"you'll\": \"you will\",\n    \"you'll've\": \"you will have\",\n    \"you're\": \"you are\",\n    \"you've\": \"you have\"\n    }\n\n    q_decontracted = []\n\n    for word in q.split():\n        if word in contractions:\n            word = contractions[word]\n\n        q_decontracted.append(word)\n\n    q = ' '.join(q_decontracted)\n    q = q.replace(\"'ve\", \" have\")\n    q = q.replace(\"n't\", \" not\")\n    q = q.replace(\"'re\", \" are\")\n    q = q.replace(\"'ll\", \" will\")\n    \n        # Remove punctuations\n    pattern = re.compile('\\W')\n    q = re.sub(pattern, ' ', q).strip()\n\n    \n    return q","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:21.781308Z","iopub.execute_input":"2022-07-10T10:10:21.781711Z","iopub.status.idle":"2022-07-10T10:10:21.801395Z","shell.execute_reply.started":"2022-07-10T10:10:21.781672Z","shell.execute_reply":"2022-07-10T10:10:21.800390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['discourse_text'] = train_df['discourse_text'].apply(lambda x: preprocess(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:21.802924Z","iopub.execute_input":"2022-07-10T10:10:21.803562Z","iopub.status.idle":"2022-07-10T10:10:23.250801Z","shell.execute_reply.started":"2022-07-10T10:10:21.803514Z","shell.execute_reply":"2022-07-10T10:10:23.249840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ntrain_df['discourse_effectiveness'] = label_encoder.fit_transform(train_df['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.252340Z","iopub.execute_input":"2022-07-10T10:10:23.252703Z","iopub.status.idle":"2022-07-10T10:10:23.271269Z","shell.execute_reply.started":"2022-07-10T10:10:23.252667Z","shell.execute_reply":"2022-07-10T10:10:23.270317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_type = list(set(train_df['discourse_type'].values))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.273191Z","iopub.execute_input":"2022-07-10T10:10:23.273463Z","iopub.status.idle":"2022-07-10T10:10:23.281213Z","shell.execute_reply.started":"2022-07-10T10:10:23.273439Z","shell.execute_reply":"2022-07-10T10:10:23.280284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse_type","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.282589Z","iopub.execute_input":"2022-07-10T10:10:23.283438Z","iopub.status.idle":"2022-07-10T10:10:23.295684Z","shell.execute_reply.started":"2022-07-10T10:10:23.283401Z","shell.execute_reply":"2022-07-10T10:10:23.294764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['text'] = train_df[\"discourse_text\"] +' '+  train_df[\"discourse_type\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.297401Z","iopub.execute_input":"2022-07-10T10:10:23.297769Z","iopub.status.idle":"2022-07-10T10:10:23.326871Z","shell.execute_reply.started":"2022-07-10T10:10:23.297720Z","shell.execute_reply":"2022-07-10T10:10:23.325985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.328764Z","iopub.execute_input":"2022-07-10T10:10:23.329385Z","iopub.status.idle":"2022-07-10T10:10:23.343802Z","shell.execute_reply.started":"2022-07-10T10:10:23.329348Z","shell.execute_reply":"2022-07-10T10:10:23.342837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_df['discourse_effectiveness']","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.345267Z","iopub.execute_input":"2022-07-10T10:10:23.345673Z","iopub.status.idle":"2022-07-10T10:10:23.350511Z","shell.execute_reply.started":"2022-07-10T10:10:23.345637Z","shell.execute_reply":"2022-07-10T10:10:23.349541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['discourse_text','discourse_id','essay_id','discourse_type' ],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.353044Z","iopub.execute_input":"2022-07-10T10:10:23.353467Z","iopub.status.idle":"2022-07-10T10:10:23.369182Z","shell.execute_reply.started":"2022-07-10T10:10:23.353432Z","shell.execute_reply":"2022-07-10T10:10:23.368188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df['text']","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.370710Z","iopub.execute_input":"2022-07-10T10:10:23.371364Z","iopub.status.idle":"2022-07-10T10:10:23.375662Z","shell.execute_reply.started":"2022-07-10T10:10:23.371329Z","shell.execute_reply":"2022-07-10T10:10:23.374730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y,\n                                                   test_size = 0.2, random_state = 4)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.377138Z","iopub.execute_input":"2022-07-10T10:10:23.377707Z","iopub.status.idle":"2022-07-10T10:10:23.389689Z","shell.execute_reply.started":"2022-07-10T10:10:23.377672Z","shell.execute_reply":"2022-07-10T10:10:23.388781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfv = TfidfVectorizer(min_df=3,  max_features=None, \n            strip_accents='unicode', analyzer='word',token_pattern=r'\\w{1,}',\n            ngram_range=(1, 3), use_idf=1,smooth_idf=1,sublinear_tf=1,\n            stop_words = 'english')\n# Fitting TF-IDF to both training and test sets (semi-supervised learning)\ntfv.fit(list(X_train) + list(X_test))\nxtrain_tfv =  tfv.transform(X_train) \nxvalid_tfv = tfv.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:23.391436Z","iopub.execute_input":"2022-07-10T10:10:23.391839Z","iopub.status.idle":"2022-07-10T10:10:30.029042Z","shell.execute_reply.started":"2022-07-10T10:10:23.391781Z","shell.execute_reply":"2022-07-10T10:10:30.027937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.multioutput import MultiOutputClassifier, MultiOutputRegressor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:30.030559Z","iopub.execute_input":"2022-07-10T10:10:30.031261Z","iopub.status.idle":"2022-07-10T10:10:30.143944Z","shell.execute_reply.started":"2022-07-10T10:10:30.031219Z","shell.execute_reply":"2022-07-10T10:10:30.142980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:30.145268Z","iopub.execute_input":"2022-07-10T10:10:30.145707Z","iopub.status.idle":"2022-07-10T10:10:30.154156Z","shell.execute_reply.started":"2022-07-10T10:10:30.145669Z","shell.execute_reply":"2022-07-10T10:10:30.153085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nxgb = XGBClassifier()\nxgb.fit(xtrain_tfv,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:10:30.163032Z","iopub.execute_input":"2022-07-10T10:10:30.163760Z","iopub.status.idle":"2022-07-10T10:12:10.432908Z","shell.execute_reply.started":"2022-07-10T10:10:30.163734Z","shell.execute_reply":"2022-07-10T10:12:10.431934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = xgb.predict(xvalid_tfv)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.435682Z","iopub.execute_input":"2022-07-10T10:12:10.436113Z","iopub.status.idle":"2022-07-10T10:12:10.554312Z","shell.execute_reply.started":"2022-07-10T10:12:10.436087Z","shell.execute_reply":"2022-07-10T10:12:10.552847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nxgb_score = accuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.558421Z","iopub.execute_input":"2022-07-10T10:12:10.559118Z","iopub.status.idle":"2022-07-10T10:12:10.569768Z","shell.execute_reply.started":"2022-07-10T10:12:10.559072Z","shell.execute_reply":"2022-07-10T10:12:10.568859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(xgb_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.571494Z","iopub.execute_input":"2022-07-10T10:12:10.571940Z","iopub.status.idle":"2022-07-10T10:12:10.577952Z","shell.execute_reply.started":"2022-07-10T10:12:10.571902Z","shell.execute_reply":"2022-07-10T10:12:10.576374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#BOW approach","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.579681Z","iopub.execute_input":"2022-07-10T10:12:10.581850Z","iopub.status.idle":"2022-07-10T10:12:10.589876Z","shell.execute_reply.started":"2022-07-10T10:12:10.581814Z","shell.execute_reply":"2022-07-10T10:12:10.588704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ncv = CountVectorizer(max_features=300)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.591605Z","iopub.execute_input":"2022-07-10T10:12:10.591965Z","iopub.status.idle":"2022-07-10T10:12:10.600329Z","shell.execute_reply.started":"2022-07-10T10:12:10.591931Z","shell.execute_reply":"2022-07-10T10:12:10.599264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv.fit(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:10.602152Z","iopub.execute_input":"2022-07-10T10:12:10.602802Z","iopub.status.idle":"2022-07-10T10:12:11.980771Z","shell.execute_reply.started":"2022-07-10T10:12:10.602762Z","shell.execute_reply":"2022-07-10T10:12:11.979766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain_cv = cv.transform(X_train).toarray()\nxtest_cv = cv.transform(X_test).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:11.982295Z","iopub.execute_input":"2022-07-10T10:12:11.982649Z","iopub.status.idle":"2022-07-10T10:12:13.759592Z","shell.execute_reply.started":"2022-07-10T10:12:11.982612Z","shell.execute_reply":"2022-07-10T10:12:13.758351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nxgb = XGBClassifier()\nxgb.fit(xtrain_cv,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:12:13.761013Z","iopub.execute_input":"2022-07-10T10:12:13.761406Z","iopub.status.idle":"2022-07-10T10:13:27.069678Z","shell.execute_reply.started":"2022-07-10T10:12:13.761366Z","shell.execute_reply":"2022-07-10T10:13:27.068742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_cv = xgb.predict(xtest_cv)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.071050Z","iopub.execute_input":"2022-07-10T10:13:27.071395Z","iopub.status.idle":"2022-07-10T10:13:27.138446Z","shell.execute_reply.started":"2022-07-10T10:13:27.071359Z","shell.execute_reply":"2022-07-10T10:13:27.137432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_score_cv = accuracy_score(y_test, y_pred_cv)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.139659Z","iopub.execute_input":"2022-07-10T10:13:27.140012Z","iopub.status.idle":"2022-07-10T10:13:27.148876Z","shell.execute_reply.started":"2022-07-10T10:13:27.139977Z","shell.execute_reply":"2022-07-10T10:13:27.147141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(xgb_score_cv)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.151628Z","iopub.execute_input":"2022-07-10T10:13:27.152178Z","iopub.status.idle":"2022-07-10T10:13:27.160793Z","shell.execute_reply.started":"2022-07-10T10:13:27.152139Z","shell.execute_reply":"2022-07-10T10:13:27.160012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB\nmodel1 = MultinomialNB().fit(xtrain_tfv,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.161878Z","iopub.execute_input":"2022-07-10T10:13:27.163414Z","iopub.status.idle":"2022-07-10T10:13:27.188896Z","shell.execute_reply.started":"2022-07-10T10:13:27.163381Z","shell.execute_reply":"2022-07-10T10:13:27.187947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred1 = model1.predict(xvalid_tfv)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.190254Z","iopub.execute_input":"2022-07-10T10:13:27.190707Z","iopub.status.idle":"2022-07-10T10:13:27.200254Z","shell.execute_reply.started":"2022-07-10T10:13:27.190665Z","shell.execute_reply":"2022-07-10T10:13:27.199228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\nprint(accuracy_score(y_test,y_pred1))\nprint(classification_report(y_pred1,y_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.201412Z","iopub.execute_input":"2022-07-10T10:13:27.201668Z","iopub.status.idle":"2022-07-10T10:13:27.224340Z","shell.execute_reply.started":"2022-07-10T10:13:27.201645Z","shell.execute_reply":"2022-07-10T10:13:27.223327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.226086Z","iopub.execute_input":"2022-07-10T10:13:27.226423Z","iopub.status.idle":"2022-07-10T10:13:27.234015Z","shell.execute_reply.started":"2022-07-10T10:13:27.226390Z","shell.execute_reply":"2022-07-10T10:13:27.232855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\ntokenizer = Tokenizer()\ntokenizer.fit_on_texts(X)\n\nX = tokenizer.texts_to_sequences(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:27.235490Z","iopub.execute_input":"2022-07-10T10:13:27.235982Z","iopub.status.idle":"2022-07-10T10:13:29.649749Z","shell.execute_reply.started":"2022-07-10T10:13:27.235955Z","shell.execute_reply":"2022-07-10T10:13:29.648642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"maxlen = 500","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:29.651434Z","iopub.execute_input":"2022-07-10T10:13:29.651860Z","iopub.status.idle":"2022-07-10T10:13:29.657301Z","shell.execute_reply.started":"2022-07-10T10:13:29.651790Z","shell.execute_reply":"2022-07-10T10:13:29.656286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\nX = pad_sequences(X, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:29.658825Z","iopub.execute_input":"2022-07-10T10:13:29.659367Z","iopub.status.idle":"2022-07-10T10:13:29.936996Z","shell.execute_reply.started":"2022-07-10T10:13:29.659332Z","shell.execute_reply":"2022-07-10T10:13:29.935978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index = tokenizer.word_index\nvocab_size = len(tokenizer.word_index) + 1","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:29.938598Z","iopub.execute_input":"2022-07-10T10:13:29.938975Z","iopub.status.idle":"2022-07-10T10:13:29.945774Z","shell.execute_reply.started":"2022-07-10T10:13:29.938938Z","shell.execute_reply":"2022-07-10T10:13:29.944857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_weights_matrix(model,vocab,) :\n    weights_matrix = np.zeros((vocab_size, DIM))\n    for word, i in vocab.items():\n        if word in list(model.wv.key_to_index):\n            weights_matrix[i] = model.wv[word]\n        \n    return weights_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:13:29.947890Z","iopub.execute_input":"2022-07-10T10:13:29.948259Z","iopub.status.idle":"2022-07-10T10:13:29.957154Z","shell.execute_reply.started":"2022-07-10T10:13:29.948224Z","shell.execute_reply":"2022-07-10T10:13:29.956126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Train test split\nX_train, X_test, y_train, y_test = train_test_split(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:14:31.807082Z","iopub.execute_input":"2022-07-10T10:14:31.807524Z","iopub.status.idle":"2022-07-10T10:14:31.853492Z","shell.execute_reply.started":"2022-07-10T10:14:31.807480Z","shell.execute_reply":"2022-07-10T10:14:31.852446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in f:\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:17:50.537063Z","iopub.execute_input":"2022-07-10T10:17:50.537712Z","iopub.status.idle":"2022-07-10T10:22:02.681521Z","shell.execute_reply.started":"2022-07-10T10:17:50.537675Z","shell.execute_reply":"2022-07-10T10:22:02.680466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in word_index.items():\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:22:02.683647Z","iopub.execute_input":"2022-07-10T10:22:02.684262Z","iopub.status.idle":"2022-07-10T10:22:02.821240Z","shell.execute_reply.started":"2022-07-10T10:22:02.684225Z","shell.execute_reply":"2022-07-10T10:22:02.815854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping\ncallbacks = [EarlyStopping(monitor='val_loss', patience=5, verbose=1),\n                ModelCheckpoint('model.hdf5',\n                                 save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:22:02.822945Z","iopub.execute_input":"2022-07-10T10:22:02.823601Z","iopub.status.idle":"2022-07-10T10:22:02.829623Z","shell.execute_reply.started":"2022-07-10T10:22:02.823544Z","shell.execute_reply":"2022-07-10T10:22:02.828554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # detect and init the TPU\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# # instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\n# # instantiating the model in the strategy scope creates the model on the TPU\n# with tpu_strategy.scope():\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,\n             300,\n             weights=[embedding_matrix],\n             input_length=1500,\n             trainable=True))\nmodel.add(Bidirectional(LSTM(300, dropout=0.3, recurrent_dropout=0.3)))\n\nmodel.add(Dense(64, activation='relu'))\n\nmodel.add(Dense(3,activation='softmax'))\nmodel.compile(loss='sparse_categorical_crossentropy', optimizer='adam',metrics=['accuracy'],)\n\n# train model normally\nmodel.fit(X_train, y_train, validation_split=0.3, batch_size = 128, epochs=6,callbacks = callbacks)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T10:22:02.833645Z","iopub.execute_input":"2022-07-10T10:22:02.834598Z","iopub.status.idle":"2022-07-10T11:35:46.686281Z","shell.execute_reply.started":"2022-07-10T10:22:02.834552Z","shell.execute_reply":"2022-07-10T11:35:46.684728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:46.689034Z","iopub.execute_input":"2022-07-10T11:35:46.689961Z","iopub.status.idle":"2022-07-10T11:35:46.708008Z","shell.execute_reply.started":"2022-07-10T11:35:46.689920Z","shell.execute_reply":"2022-07-10T11:35:46.706837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_df['discourse_text']+' '+ test_df['discourse_type']","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:46.709584Z","iopub.execute_input":"2022-07-10T11:35:46.710590Z","iopub.status.idle":"2022-07-10T11:35:46.720007Z","shell.execute_reply.started":"2022-07-10T11:35:46.710544Z","shell.execute_reply":"2022-07-10T11:35:46.718847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = tokenizer.texts_to_sequences(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:46.722235Z","iopub.execute_input":"2022-07-10T11:35:46.723605Z","iopub.status.idle":"2022-07-10T11:35:46.731572Z","shell.execute_reply.started":"2022-07-10T11:35:46.723558Z","shell.execute_reply":"2022-07-10T11:35:46.730457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pad_sequences(test_data, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:46.733727Z","iopub.execute_input":"2022-07-10T11:35:46.734804Z","iopub.status.idle":"2022-07-10T11:35:46.742133Z","shell.execute_reply.started":"2022-07-10T11:35:46.734602Z","shell.execute_reply":"2022-07-10T11:35:46.741010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:46.744189Z","iopub.execute_input":"2022-07-10T11:35:46.745032Z","iopub.status.idle":"2022-07-10T11:35:47.538685Z","shell.execute_reply.started":"2022-07-10T11:35:46.744989Z","shell.execute_reply":"2022-07-10T11:35:47.537552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_sample.loc[:,\"Ineffective\"] = pred[:,0]\nsubmission_sample.loc[:,\"Adequate\"] = pred[:,1]\nsubmission_sample.loc[:,\"Effective\"] = pred[:,2]\nsubmission_sample.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:39:53.355438Z","iopub.execute_input":"2022-07-10T11:39:53.356938Z","iopub.status.idle":"2022-07-10T11:39:53.367391Z","shell.execute_reply.started":"2022-07-10T11:39:53.356879Z","shell.execute_reply":"2022-07-10T11:39:53.366150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:35:47.656127Z","iopub.execute_input":"2022-07-10T11:35:47.656943Z","iopub.status.idle":"2022-07-10T11:35:47.674262Z","shell.execute_reply.started":"2022-07-10T11:35:47.656898Z","shell.execute_reply":"2022-07-10T11:35:47.673146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}