{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Load the dataset from kaggle's data directory.\nCheck the working directory of kaggle's dataset and load its train dataset.","metadata":{}},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/feedback-prize-effectiveness')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:20.397338Z","iopub.execute_input":"2022-08-10T07:07:20.397965Z","iopub.status.idle":"2022-08-10T07:07:20.429170Z","shell.execute_reply.started":"2022-08-10T07:07:20.397854Z","shell.execute_reply":"2022-08-10T07:07:20.428255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n\ndf_train = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ndf_test = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\ndf_submission = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\ndf_train['discourse_text'].isnull().sum() # checks for NAs\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:20.431269Z","iopub.execute_input":"2022-08-10T07:07:20.431929Z","iopub.status.idle":"2022-08-10T07:07:21.272509Z","shell.execute_reply.started":"2022-08-10T07:07:20.431893Z","shell.execute_reply":"2022-08-10T07:07:21.271299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data pre-processing\n\nRemove stopwords from corpus stopwords dictionary to prevent commonly used language or text such as 'a', 'the', etc, that may affect the predictive performance.","metadata":{}},{"cell_type":"code","source":"import string\nstring.punctuation \nimport re\nfrom nltk.corpus import stopwords \nfrom collections import Counter\nfrom itertools import chain\n\nstop = set(stopwords.words('english')) \nprint(stop)\ndf_train['discourse_text'].replace(\"[^a-zA-Z]\",\" \", regex=True, inplace =True)  # match strings that contains non-letter and replace with black to remove string.punctuations.\ndf_train['discourse_text'] = df_train['discourse_text'].str.lower() # converts strings to lower case.\nprint(df_train)\ndf_train['discourse_text'] = df_train['discourse_text'].apply(lambda x: [item for item in str(x).split() if item not in stop])\nprint(df_train['discourse_text'])\nfreq = pd.Series(Counter(chain.from_iterable(df_train['discourse_text']))).sort_values(ascending=False).reset_index() # count the frequencies of words.\nprint(freq)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:21.277074Z","iopub.execute_input":"2022-08-10T07:07:21.280561Z","iopub.status.idle":"2022-08-10T07:07:22.749440Z","shell.execute_reply.started":"2022-08-10T07:07:21.280521Z","shell.execute_reply":"2022-08-10T07:07:22.748392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Note: To access and run nltk corpora 'owm-1.4' file without internet, simply download the file at https://raw.githubusercontent.com/nltk/nltk_data/gh-pages/packages/corpora/omw-1.4.zip , upload to kaggle dataset and run the below command to install the file from kaggle input into the kernel.","metadata":{}},{"cell_type":"code","source":"!ln -s /kaggle/input/nltk-omw/omw-1.4 /usr/share/nltk_data/corpora/omw-1.4 ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:22.752101Z","iopub.execute_input":"2022-08-10T07:07:22.752729Z","iopub.status.idle":"2022-08-10T07:07:23.728132Z","shell.execute_reply.started":"2022-08-10T07:07:22.752692Z","shell.execute_reply":"2022-08-10T07:07:23.726839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### To test if the file is working:","metadata":{}},{"cell_type":"code","source":"import nltk\nnltk.corpus.wordnet.synsets('think')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:23.730002Z","iopub.execute_input":"2022-08-10T07:07:23.730971Z","iopub.status.idle":"2022-08-10T07:07:25.662150Z","shell.execute_reply.started":"2022-08-10T07:07:23.730938Z","shell.execute_reply":"2022-08-10T07:07:25.661167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Lemmatize all tokens into a new list to prevent overfitting of dataset when verb, nouns, adverb or adjectives are not as concerning on the impact of effectiveness of an argument.\n","metadata":{}},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\nfrom nltk import word_tokenize\nwordnet_lem = WordNetLemmatizer()\ndf_train['discourse_text'] = df_train['discourse_text'].apply(\n                   lambda lst:[wordnet_lem.lemmatize(word, pos='v') for word in lst])\ndf_train['discourse_text'] = df_train['discourse_text'].apply(\n                    lambda lst:[wordnet_lem.lemmatize(word, pos='n') for word in lst])\ndf_train['discourse_text'] = df_train['discourse_text'].apply(\n                    lambda lst:[wordnet_lem.lemmatize(word, pos='r') for word in lst])\ndf_train['discourse_text'] = df_train['discourse_text'].apply(\n                    lambda lst:[wordnet_lem.lemmatize(word, pos='a') for word in lst])\nfreq_final = pd.Series(Counter(chain.from_iterable(df_train['discourse_text']))).sort_values(ascending=False)\nprint(freq_final)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:25.663612Z","iopub.execute_input":"2022-08-10T07:07:25.664157Z","iopub.status.idle":"2022-08-10T07:07:39.534406Z","shell.execute_reply.started":"2022-08-10T07:07:25.664120Z","shell.execute_reply":"2022-08-10T07:07:39.533387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Check the number of repeating words.","metadata":{}},{"cell_type":"code","source":"repeating_words = len([1 for i in freq_final if i >1])\nprint(repeating_words)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:39.536134Z","iopub.execute_input":"2022-08-10T07:07:39.536838Z","iopub.status.idle":"2022-08-10T07:07:39.547710Z","shell.execute_reply.started":"2022-08-10T07:07:39.536800Z","shell.execute_reply":"2022-08-10T07:07:39.546172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualize top 30 word frequencies. ","metadata":{}},{"cell_type":"code","source":"import seaborn as sns \nimport matplotlib.pyplot as plt\ntop_words = freq_final.head(30).reset_index()\n\nsns.reset_orig()\nplt.figure(figsize = (8,6))\nmy_palette = sns.color_palette(\"colorblind\") # variations of default palette: deep, muted, pastel, bright, dark, colorblind. \nplt.style.use('seaborn-colorblind')\nsns.set(rc={'figure.figsize':(10,5)})\nsns.barplot(x=top_words.iloc[:,1], y=top_words.iloc[:,0],data=top_words, alpha = 0.6).set_title('Top 20 count of word frequencies')\nplt.xlabel('Frequencies')\nplt.ylabel('Words')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:39.549281Z","iopub.execute_input":"2022-08-10T07:07:39.549573Z","iopub.status.idle":"2022-08-10T07:07:40.124088Z","shell.execute_reply.started":"2022-08-10T07:07:39.549531Z","shell.execute_reply":"2022-08-10T07:07:40.121823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualize the dataset to determine if data is normal distributed.","metadata":{}},{"cell_type":"code","source":"counts = df_train.discourse_effectiveness.value_counts()\nprint(counts)\nprint(\"\\nPredicting only 0 = {:.2f}% accuracy\".format(counts[0] / sum(counts) * 100))\nprint(\"\\nPredicting only 1 = {:.2f}% accuracy\".format(counts[1] / sum(counts) * 100))\nprint(\"\\nPredicting only 2 = {:.2f}% accuracy\".format(counts[2] / sum(counts) * 100))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:40.125781Z","iopub.execute_input":"2022-08-10T07:07:40.126149Z","iopub.status.idle":"2022-08-10T07:07:40.136919Z","shell.execute_reply.started":"2022-08-10T07:07:40.126113Z","shell.execute_reply":"2022-08-10T07:07:40.135737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts = pd.DataFrame(counts).reset_index()\nplt.figure(figsize = (10,5))\nsns.barplot(x=df_counts['index'],y=df_counts['discourse_effectiveness'],data=df_counts,alpha = 0.6).set_title('Total count of Effectiveness')\nplt.xlabel('Effectiveness')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:40.142109Z","iopub.execute_input":"2022-08-10T07:07:40.142801Z","iopub.status.idle":"2022-08-10T07:07:40.354099Z","shell.execute_reply.started":"2022-08-10T07:07:40.142765Z","shell.execute_reply":"2022-08-10T07:07:40.353158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It can be observed that there are more Adequate feedbacks than Effective and Ineffective, indicating that the data may be imbalanced and may be prone to lower prediction accuracy. ","metadata":{}},{"cell_type":"markdown","source":"#### One-hot-encoding by creating dummies to categorical data.","metadata":{}},{"cell_type":"code","source":"df_effects = pd.get_dummies(df_train.iloc[:,4])\ndf_train = pd.concat([df_train, df_effects], axis=1) # combine dummy rows.","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:40.355852Z","iopub.execute_input":"2022-08-10T07:07:40.356557Z","iopub.status.idle":"2022-08-10T07:07:40.371142Z","shell.execute_reply.started":"2022-08-10T07:07:40.356520Z","shell.execute_reply":"2022-08-10T07:07:40.369859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train-test validation approach.","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(df_train['discourse_text'].values, df_train[['Adequate', 'Effective','Ineffective']].values, stratify=df_train['discourse_effectiveness'],test_size=0.1, random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:40.372644Z","iopub.execute_input":"2022-08-10T07:07:40.373014Z","iopub.status.idle":"2022-08-10T07:07:40.410351Z","shell.execute_reply.started":"2022-08-10T07:07:40.372980Z","shell.execute_reply":"2022-08-10T07:07:40.409500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Tokenize most common words to prevent overftitting from noise text that least occur.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nmax_length = 300\n\ntokenizer = Tokenizer(num_words= repeating_words, oov_token='x')\nword_index = tokenizer.word_index\ncount_words = tokenizer.word_counts\ntokenizer.fit_on_texts(X_train) \ntokenizer.fit_on_texts(X_test)\nseq_train = tokenizer.texts_to_sequences(X_train)\nseq_test = tokenizer.texts_to_sequences(X_test)\npad_train = pad_sequences(seq_train,maxlen = max_length  ) #\npad_test = pad_sequences(seq_test,maxlen = max_length) #\n\n\npad_train.shape, pad_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:40.411639Z","iopub.execute_input":"2022-08-10T07:07:40.412072Z","iopub.status.idle":"2022-08-10T07:07:46.755376Z","shell.execute_reply.started":"2022-08-10T07:07:40.412036Z","shell.execute_reply":"2022-08-10T07:07:46.754190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = len(tokenizer.word_index)+1\nprint(vocab_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:46.757057Z","iopub.execute_input":"2022-08-10T07:07:46.758678Z","iopub.status.idle":"2022-08-10T07:07:46.767626Z","shell.execute_reply.started":"2022-08-10T07:07:46.758631Z","shell.execute_reply":"2022-08-10T07:07:46.765850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Shuffle train set after splitting to improve or avoid overfitting and ensure data are representatives.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import shuffle\npad_train, y_train = shuffle(pad_train, y_train)\nprint(pad_train[9])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:46.769609Z","iopub.execute_input":"2022-08-10T07:07:46.770456Z","iopub.status.idle":"2022-08-10T07:07:46.801965Z","shell.execute_reply.started":"2022-08-10T07:07:46.770419Z","shell.execute_reply":"2022-08-10T07:07:46.800920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating LST model to predict effectiveness of arguments.\nCreate LST model and add neuron layers to easily define relationship of the output classess. Last layer based on number of desired categorical/classes output of interest.\n\nLinear regularization used in optimizer settings to prevent overfitting of data.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import regularizers\nimport keras\nfrom keras.callbacks import EarlyStopping\nfrom tensorflow.keras.models import Sequential \nimport tensorflow\nfrom tensorflow.keras.layers import Dropout\n\n\nlst_mod = tf.keras.Sequential([\n    tf.keras.layers.Embedding(input_dim=vocab_size, input_length = max_length, output_dim=100),\n    tf.keras.layers.SpatialDropout1D(0.6),\n    tf.keras.layers.LSTM(512, dropout = 0.5),\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.Dropout(0.4),\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(3, activation='softmax')\n])\n\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nloss = CategoricalCrossentropy(from_logits = True)\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.0001)\nlst_mod.compile(loss=loss, optimizer=optimizer, metrics=['accuracy'])\nlst_mod.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:46.803340Z","iopub.execute_input":"2022-08-10T07:07:46.804295Z","iopub.status.idle":"2022-08-10T07:07:50.447016Z","shell.execute_reply.started":"2022-08-10T07:07:46.804256Z","shell.execute_reply":"2022-08-10T07:07:50.445950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Implementing early stopper to prevent overfitting of data that may occur on validation accuracy.\n","metadata":{}},{"cell_type":"code","source":"class earlystop(tf.keras.callbacks.Callback):\n  def on_epoch_end(self, epoch, logs={}): \n    if(logs.get('accuracy')>0.855):\n      print(\"Accuracy has reached > 85.5%!\") \n      self.model.stop_training = True\nes = earlystop() ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:50.448756Z","iopub.execute_input":"2022-08-10T07:07:50.449473Z","iopub.status.idle":"2022-08-10T07:07:50.456106Z","shell.execute_reply.started":"2022-08-10T07:07:50.449435Z","shell.execute_reply":"2022-08-10T07:07:50.454537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After numerous trial runs, epochs = 20 provides constant increase in test accuracy. If epochs > 20, LST model may overfit, where train accuracy graduaually increasing while validation accuracy decreases.\n\n##### Note:To run batches of data faster, it is recommended to turn on Accelerator to GPU under settings.","metadata":{}},{"cell_type":"code","source":"lst_mod1 = lst_mod.fit(pad_train, y_train, epochs=20, callbacks=[es],\n            validation_data=(pad_test, y_test), verbose=1, batch_size=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:07:50.457682Z","iopub.execute_input":"2022-08-10T07:07:50.458051Z","iopub.status.idle":"2022-08-10T07:16:14.068395Z","shell.execute_reply.started":"2022-08-10T07:07:50.458016Z","shell.execute_reply":"2022-08-10T07:16:14.067295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualize overall accuracy and validation accuracy, as well as loss and validation loss of individual epochs of the LST model.\n\n##### Note: Saving LST model into a variable allows the model to be visualized on graph, else it would return an error where history is not callable.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\nplt.plot(lst_mod1.history['accuracy'])\nplt.plot(lst_mod1.history['val_accuracy'])\nplt.title('accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:14.070415Z","iopub.execute_input":"2022-08-10T07:16:14.070798Z","iopub.status.idle":"2022-08-10T07:16:14.356168Z","shell.execute_reply.started":"2022-08-10T07:16:14.070761Z","shell.execute_reply":"2022-08-10T07:16:14.355219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\nplt.plot(lst_mod1.history['loss'])\nplt.plot(lst_mod1.history['val_loss'])\nplt.title('loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:14.357548Z","iopub.execute_input":"2022-08-10T07:16:14.358170Z","iopub.status.idle":"2022-08-10T07:16:14.636209Z","shell.execute_reply.started":"2022-08-10T07:16:14.358131Z","shell.execute_reply":"2022-08-10T07:16:14.635174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Predict LST model on padded test set from train-test split to determine the overall accuracy of the model before predicting on test.csv dataset.","metadata":{}},{"cell_type":"code","source":"y_predict = lst_mod.predict(pad_test, verbose=0)\nprint(y_predict)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:14.637879Z","iopub.execute_input":"2022-08-10T07:16:14.638574Z","iopub.status.idle":"2022-08-10T07:16:17.524484Z","shell.execute_reply.started":"2022-08-10T07:16:14.638537Z","shell.execute_reply":"2022-08-10T07:16:17.523313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Confusion matrix","metadata":{}},{"cell_type":"code","source":"import sklearn.metrics as metrics\nfrom sklearn.metrics import classification_report,confusion_matrix,accuracy_score\ntest_cm = metrics.confusion_matrix(y_test.argmax(axis=1), y_predict.argmax(axis=1))\nprint(test_cm)\ntest_score = metrics.accuracy_score(y_test.argmax(axis=1), y_predict.argmax(axis=1))\nprint(test_score)\ntest_report = metrics.classification_report(y_test.argmax(axis=1), y_predict.argmax(axis=1))\nprint(test_report)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:17.526198Z","iopub.execute_input":"2022-08-10T07:16:17.526899Z","iopub.status.idle":"2022-08-10T07:16:17.550329Z","shell.execute_reply.started":"2022-08-10T07:16:17.526862Z","shell.execute_reply":"2022-08-10T07:16:17.549447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It can be seen that basesd on classification report, 0: 'Adequate' has the highest overall accuracy in terms of precision (actual true positive ouf of predicted positive), recall (true positive rate) and f1-score (mean of precision and recall, taking consideration of false positive and false negatives)\n\nIt can also be observed that 'Adequate' has the highest number of observations supporting its accuracy.\n\nFollowed by 1:'Effective' with the second highest accuracy and 2: 'Ineffective' as the least overall accuracy. \n","metadata":{}},{"cell_type":"markdown","source":"#### Visualize Confusion Matrix","metadata":{}},{"cell_type":"code","source":"test_cm = pd.DataFrame(test_cm, range(3), range(3))\nplt.figure(figsize = (10,8))\nsns.set(font_scale=2)\nsns.heatmap(test_cm, annot=True, annot_kws={\"size\": 21},fmt='d')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:17.551646Z","iopub.execute_input":"2022-08-10T07:16:17.552117Z","iopub.status.idle":"2022-08-10T07:16:17.819079Z","shell.execute_reply.started":"2022-08-10T07:16:17.552081Z","shell.execute_reply":"2022-08-10T07:16:17.818136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Similarly, data pre-process on test.csv dataset by removing stopwords, lemmatize and tokenize text into sequences of numbers for each letters and pad accordingly to faclitate prediction using above model on text in test.csv dataset","metadata":{}},{"cell_type":"code","source":"df_test['discourse_text'].replace(\"[^a-zA-Z]\",\" \", regex=True, inplace =True)  # match strings that contains non-letter and replace with black to remove string.punctuations.\ndf_test['discourse_text'] = df_test['discourse_text'].str.lower() # converts strings to lower case.\ndf_test['discourse_text'] = df_test['discourse_text'].apply(lambda x: [item for item in str(x).split() if item not in stop])\nprint(df_test['discourse_text'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:17.820651Z","iopub.execute_input":"2022-08-10T07:16:17.821350Z","iopub.status.idle":"2022-08-10T07:16:17.835457Z","shell.execute_reply.started":"2022-08-10T07:16:17.821312Z","shell.execute_reply":"2022-08-10T07:16:17.834354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_test_index = tokenizer.word_index\ncount_test_words = tokenizer.word_counts\ntokenizer.fit_on_texts(df_test.discourse_text)\nseq_df_test = tokenizer.texts_to_sequences(df_test.discourse_text)\npad_df_test = pad_sequences(seq_df_test)\n\n\npad_df_test = shuffle(pad_df_test)\nprint(pad_df_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:17.837343Z","iopub.execute_input":"2022-08-10T07:16:17.837751Z","iopub.status.idle":"2022-08-10T07:16:17.875087Z","shell.execute_reply.started":"2022-08-10T07:16:17.837715Z","shell.execute_reply":"2022-08-10T07:16:17.873811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_predict = lst_mod.predict(pad_df_test, verbose=0)\nprint(y_test_predict)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:17.876933Z","iopub.execute_input":"2022-08-10T07:16:17.877297Z","iopub.status.idle":"2022-08-10T07:16:18.216595Z","shell.execute_reply.started":"2022-08-10T07:16:17.877261Z","shell.execute_reply":"2022-08-10T07:16:18.215430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Round to nearest deciminal places,replace submissions dataset with test predictions on test.csv, and export as 'submission.csv' output.\n","metadata":{}},{"cell_type":"code","source":"df_test_predict = pd.DataFrame(y_test_predict, columns=['Adequate','Effective','Ineffective'])\nfinal_test_dataset = pd.merge(df_test, df_test_predict, how = 'right', left_index= True ,right_index =True)\nfinal_test_dataset = final_test_dataset[['discourse_id', 'Ineffective','Adequate', 'Effective',]]\n\nfinal_test_dataset.reset_index()\ndecimals = 5    \nfinal_test_dataset.iloc[:,1:4] = final_test_dataset.iloc[:,1:4].apply(lambda x: round(x, decimals))\nprint(final_test_dataset)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:18.218310Z","iopub.execute_input":"2022-08-10T07:16:18.218665Z","iopub.status.idle":"2022-08-10T07:16:18.241026Z","shell.execute_reply.started":"2022-08-10T07:16:18.218628Z","shell.execute_reply":"2022-08-10T07:16:18.239938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.iloc[:,1] = final_test_dataset.iloc[:,1]\ndf_submission.iloc[:,2] = final_test_dataset.iloc[:,2]\ndf_submission.iloc[:,3] = final_test_dataset.iloc[:,3]\nprint(df_submission)\ndf_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:16:18.246322Z","iopub.execute_input":"2022-08-10T07:16:18.247111Z","iopub.status.idle":"2022-08-10T07:16:18.260991Z","shell.execute_reply.started":"2022-08-10T07:16:18.247077Z","shell.execute_reply":"2022-08-10T07:16:18.259893Z"},"trusted":true},"execution_count":null,"outputs":[]}]}