{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#%load_ext pycodestyle_magic","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:05.674954Z","iopub.execute_input":"2022-08-03T18:45:05.675332Z","iopub.status.idle":"2022-08-03T18:45:05.681552Z","shell.execute_reply.started":"2022-08-03T18:45:05.675301Z","shell.execute_reply":"2022-08-03T18:45:05.680301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%pycodestyle_on","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:05.687957Z","iopub.execute_input":"2022-08-03T18:45:05.688876Z","iopub.status.idle":"2022-08-03T18:45:05.694795Z","shell.execute_reply.started":"2022-08-03T18:45:05.688839Z","shell.execute_reply":"2022-08-03T18:45:05.693321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# original notebook :\n# https://www.kaggle.com/code/ravikumarmn/feedback-prize-effectiveness-eda-bert-base/notebook","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:05.696945Z","iopub.execute_input":"2022-08-03T18:45:05.698092Z","iopub.status.idle":"2022-08-03T18:45:05.708192Z","shell.execute_reply.started":"2022-08-03T18:45:05.698053Z","shell.execute_reply":"2022-08-03T18:45:05.707222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feedback Prize - Predicting Effective Arguments**","metadata":{"papermill":{"duration":0.032419,"end_time":"2022-07-20T11:17:42.175410","exception":false,"start_time":"2022-07-20T11:17:42.142991","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### **The dataset presented here contains argumentative essays written by U.S students in grades 6-12. These essays were annotated by expert raters for discourse elements commonly found in argumentative writing:**\n\n> **Lead** - an introduction that begins with a statistic, a quotation, a description, or some other device to grab the reader’s attention and point toward the thesis\n\n> **Position** - an opinion or conclusion on the main question\n\n> **Claim** - a claim that supports the position\n\n> **Counterclaim** - a claim that refutes another claim or gives an opposing reason to the position\n\n> **Rebuttal** - a claim that refutes a counterclaim\n\n> **Evidence** - ideas or examples that support claims, counterclaims, or rebuttals.\n\n> **Concluding Statement** - a concluding statement that restates the claims\n\n#### **Your task is to predict the quality rating of each discourse element. Human readers rated each rhetorical or argumentative element, in order of increasing quality, as one of:**\n\n> **Ineffective**\n\n> **Adequate**\n\n> **Effective**\n","metadata":{"papermill":{"duration":0.028893,"end_time":"2022-07-20T11:17:42.235221","exception":false,"start_time":"2022-07-20T11:17:42.206328","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# **Importing Libraries:** ","metadata":{"papermill":{"duration":0.027876,"end_time":"2022-07-20T11:17:42.291207","exception":false,"start_time":"2022-07-20T11:17:42.263331","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy\nfrom keras.utils.vis_utils import plot_model\n\nimport transformers\nfrom transformers import TFBertModel\nfrom transformers import AutoTokenizer\n\nimport warnings\nfrom tqdm import tqdm\n\npd.set_option(\"max_colwidth\", None)\nplt.rcParams.update({'font.size': 14})\nplt.rc('legend', fontsize=10)\n\nwarnings.filterwarnings('ignore')\ntf.config.experimental_run_functions_eagerly(False)\n\ntqdm.pandas()","metadata":{"_kg_hide-output":true,"papermill":{"duration":12.524119,"end_time":"2022-07-20T11:17:54.843873","exception":false,"start_time":"2022-07-20T11:17:42.319754","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:05.709639Z","iopub.execute_input":"2022-08-03T18:45:05.710762Z","iopub.status.idle":"2022-08-03T18:45:21.314297Z","shell.execute_reply.started":"2022-08-03T18:45:05.710688Z","shell.execute_reply":"2022-08-03T18:45:21.313288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading Dataset :**","metadata":{"papermill":{"duration":0.029229,"end_time":"2022-07-20T11:17:54.903856","exception":false,"start_time":"2022-07-20T11:17:54.874627","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def print_size(df, label):\n    \"\"\"Print information about dataframe `df` size (rows x columns)\"\"\"\n    shape = df.shape\n    print(f\"{label.upper()}: {shape[0]} rows x {shape[1]} columns\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:21.317211Z","iopub.execute_input":"2022-08-03T18:45:21.318167Z","iopub.status.idle":"2022-08-03T18:45:21.326724Z","shell.execute_reply.started":"2022-08-03T18:45:21.318127Z","shell.execute_reply":"2022-08-03T18:45:21.325632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntest_data = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsubmission_data = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")","metadata":{"papermill":{"duration":0.311565,"end_time":"2022-07-20T11:17:55.245262","exception":false,"start_time":"2022-07-20T11:17:54.933697","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:21.328282Z","iopub.execute_input":"2022-08-03T18:45:21.328926Z","iopub.status.idle":"2022-08-03T18:45:21.619575Z","shell.execute_reply.started":"2022-08-03T18:45:21.328883Z","shell.execute_reply":"2022-08-03T18:45:21.618638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_size(train_data, 'train')\nprint_size(test_data, 'test')\nprint_size(submission_data, 'submission')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:21.620954Z","iopub.execute_input":"2022-08-03T18:45:21.621329Z","iopub.status.idle":"2022-08-03T18:45:21.631082Z","shell.execute_reply.started":"2022-08-03T18:45:21.621290Z","shell.execute_reply":"2022-08-03T18:45:21.629935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Exploratory Data Analysis :**","metadata":{"papermill":{"duration":0.029823,"end_time":"2022-07-20T11:17:55.305911","exception":false,"start_time":"2022-07-20T11:17:55.276088","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_data.head()","metadata":{"papermill":{"duration":0.056143,"end_time":"2022-07-20T11:17:55.391372","exception":false,"start_time":"2022-07-20T11:17:55.335229","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:21.633014Z","iopub.execute_input":"2022-08-03T18:45:21.633480Z","iopub.status.idle":"2022-08-03T18:45:21.656502Z","shell.execute_reply.started":"2022-08-03T18:45:21.633445Z","shell.execute_reply":"2022-08-03T18:45:21.655619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Types of discourse ","metadata":{"papermill":{"duration":0.029933,"end_time":"2022-07-20T11:17:55.451314","exception":false,"start_time":"2022-07-20T11:17:55.421381","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"DISCOURSE TYPES :  {train_data.discourse_type.unique().tolist()}\")","metadata":{"papermill":{"duration":0.043114,"end_time":"2022-07-20T11:17:55.524570","exception":false,"start_time":"2022-07-20T11:17:55.481456","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:21.658004Z","iopub.execute_input":"2022-08-03T18:45:21.658375Z","iopub.status.idle":"2022-08-03T18:45:21.673103Z","shell.execute_reply.started":"2022-08-03T18:45:21.658340Z","shell.execute_reply":"2022-08-03T18:45:21.671952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Types of effective discourse ","metadata":{"papermill":{"duration":0.029341,"end_time":"2022-07-20T11:17:55.583415","exception":false,"start_time":"2022-07-20T11:17:55.554074","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"EFFECTIVE DISCOURSE : {train_data.discourse_effectiveness.unique()}\")","metadata":{"papermill":{"duration":0.04072,"end_time":"2022-07-20T11:17:55.654998","exception":false,"start_time":"2022-07-20T11:17:55.614278","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:21.674910Z","iopub.execute_input":"2022-08-03T18:45:21.675306Z","iopub.status.idle":"2022-08-03T18:45:21.688593Z","shell.execute_reply.started":"2022-08-03T18:45:21.675262Z","shell.execute_reply":"2022-08-03T18:45:21.687660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Total discourse type count ","metadata":{"papermill":{"duration":0.029918,"end_time":"2022-07-20T11:17:55.715578","exception":false,"start_time":"2022-07-20T11:17:55.685660","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_axis = train_data.discourse_type.value_counts().index\ny_axis = train_data.discourse_type.value_counts().values","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.803454,"end_time":"2022-07-20T11:17:56.549570","exception":false,"start_time":"2022-07-20T11:17:55.746116","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:21.695359Z","iopub.execute_input":"2022-08-03T18:45:21.695654Z","iopub.status.idle":"2022-08-03T18:45:21.708900Z","shell.execute_reply.started":"2022-08-03T18:45:21.695629Z","shell.execute_reply":"2022-08-03T18:45:21.707957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(x=x_axis, y=y_axis,\n             color=x_axis,\n             title='Total count of discourse types in dataset ',\n             labels=dict(x='Discourse types', y='Total discourse type count'))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:21.710314Z","iopub.execute_input":"2022-08-03T18:45:21.710702Z","iopub.status.idle":"2022-08-03T18:45:22.506443Z","shell.execute_reply.started":"2022-08-03T18:45:21.710666Z","shell.execute_reply":"2022-08-03T18:45:22.505536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Total count of discourse effectiveness","metadata":{"papermill":{"duration":0.031503,"end_time":"2022-07-20T11:17:56.612048","exception":false,"start_time":"2022-07-20T11:17:56.580545","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_axis = train_data.discourse_effectiveness.value_counts().index\ny_axis = train_data.discourse_effectiveness.value_counts().values\nfig = px.bar(x=x_axis, y=y_axis,\n             color=x_axis,\n             title='Total count of dicourse effectiveness',\n             labels=dict(x='Discourse effectivenss',\n                         y='Count of effective discourse'))\nfig.show()","metadata":{"papermill":{"duration":0.11514,"end_time":"2022-07-20T11:17:56.758699","exception":false,"start_time":"2022-07-20T11:17:56.643559","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:22.508079Z","iopub.execute_input":"2022-08-03T18:45:22.508580Z","iopub.status.idle":"2022-08-03T18:45:22.579807Z","shell.execute_reply.started":"2022-08-03T18:45:22.508543Z","shell.execute_reply":"2022-08-03T18:45:22.578903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count of discourse effectiveness with respect to discourse types","metadata":{"papermill":{"duration":0.031261,"end_time":"2022-07-20T11:17:56.821716","exception":false,"start_time":"2022-07-20T11:17:56.790455","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def plot_normalized_stacked_bars(df, var1, var2):\n    \"\"\"Get dataframe with two columns and plot\n    crosstab of those columns\"\"\"\n    crosstab = pd.crosstab(\n        df[var1],\n        df[var2])\n\n    for idx, row in crosstab.iterrows():\n        crosstab.loc[idx, :] = (100 * row / sum(row))\n\n    crosstab_long = crosstab.unstack()\n    crosstab_long.name = \"%\"\n    crosstab_long = crosstab_long.reset_index()\n\n    return px.bar(\n        crosstab_long,\n        x='%', y=var1, color=var2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:22.581359Z","iopub.execute_input":"2022-08-03T18:45:22.581751Z","iopub.status.idle":"2022-08-03T18:45:22.593575Z","shell.execute_reply.started":"2022-08-03T18:45:22.581713Z","shell.execute_reply":"2022-08-03T18:45:22.592668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_normalized_stacked_bars(\n    train_data, 'discourse_type', 'discourse_effectiveness')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:22.595278Z","iopub.execute_input":"2022-08-03T18:45:22.595690Z","iopub.status.idle":"2022-08-03T18:45:22.691316Z","shell.execute_reply.started":"2022-08-03T18:45:22.595654Z","shell.execute_reply":"2022-08-03T18:45:22.690445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=1, ncols=1, figsize=(22, 10))\neffectiveness_order = ['Ineffective', 'Adequate', 'Effective']\neffectiveness_colors = ['lightgreen', 'blue', 'red']\nsns.countplot(data=train_data, x='discourse_type',\n              hue='discourse_effectiveness',\n              hue_order=effectiveness_order,\n              palette=effectiveness_colors)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.426093,"end_time":"2022-07-20T11:17:57.283480","exception":false,"start_time":"2022-07-20T11:17:56.857387","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:22.692716Z","iopub.execute_input":"2022-08-03T18:45:22.693156Z","iopub.status.idle":"2022-08-03T18:45:23.052483Z","shell.execute_reply.started":"2022-08-03T18:45:22.693120Z","shell.execute_reply":"2022-08-03T18:45:23.051536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Histogram of train word counts","metadata":{"papermill":{"duration":0.034342,"end_time":"2022-07-20T11:17:57.352547","exception":false,"start_time":"2022-07-20T11:17:57.318205","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_token_count(x):\n    \"\"\"Return number of tokens in given sentence\"\"\"\n    return len(x.split())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:23.053860Z","iopub.execute_input":"2022-08-03T18:45:23.054315Z","iopub.status.idle":"2022-08-03T18:45:23.062055Z","shell.execute_reply.started":"2022-08-03T18:45:23.054278Z","shell.execute_reply":"2022-08-03T18:45:23.060981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"discourse_num_words\"] = train_data.discourse_text.apply(get_token_count)  # ignore line too long  # noqa\n\nplt.hist(train_data[\"discourse_num_words\"], bins=100)\nplt.title('Histogram of Train Word Counts', size=16)\nplt.xlabel('Train Word Count', size=14)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.48512,"end_time":"2022-07-20T11:17:57.871896","exception":false,"start_time":"2022-07-20T11:17:57.386776","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.064167Z","iopub.execute_input":"2022-08-03T18:45:23.065233Z","iopub.status.idle":"2022-08-03T18:45:23.464279Z","shell.execute_reply.started":"2022-08-03T18:45:23.065196Z","shell.execute_reply":"2022-08-03T18:45:23.463419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing** ","metadata":{"papermill":{"duration":0.03343,"end_time":"2022-07-20T11:17:57.938709","exception":false,"start_time":"2022-07-20T11:17:57.905279","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Bert Encoder","metadata":{"papermill":{"duration":0.035839,"end_time":"2022-07-20T11:17:58.008110","exception":false,"start_time":"2022-07-20T11:17:57.972271","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def bert_encoder(texts, tokenizer, max_len=256):\n    \"\"\"Take `texts`, `tokenizer`, and `max_len` and tokenize texts\n    with given tokenizer.\n\n    Return `input_ids`, `token_type_ids`, `attention_mask` suitable for\n    `model.fit` \"\"\"\n    input_ids = list()\n    token_type_ids = list()\n    attention_mask = list()\n\n    for text in texts:\n        token = tokenizer(\n            text, max_length=256,\n            truncation=True, padding='max_length',\n            add_special_tokens=True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n\n    input_ids = np.array(input_ids)\n    token_type_ids = np.array(token_type_ids)\n    attention_mask = np.array(attention_mask)\n\n    return input_ids, token_type_ids, attention_mask","metadata":{"papermill":{"duration":0.042369,"end_time":"2022-07-20T11:17:58.084048","exception":false,"start_time":"2022-07-20T11:17:58.041679","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.465574Z","iopub.execute_input":"2022-08-03T18:45:23.466759Z","iopub.status.idle":"2022-08-03T18:45:23.478695Z","shell.execute_reply.started":"2022-08-03T18:45:23.466721Z","shell.execute_reply":"2022-08-03T18:45:23.477734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bert Tokenizer","metadata":{"papermill":{"duration":0.03307,"end_time":"2022-07-20T11:17:58.150609","exception":false,"start_time":"2022-07-20T11:17:58.117539","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Bert Tokenizer\nmodel_path = '../input/bertbasecased/bert-base-cased'\ntokenizer = AutoTokenizer.from_pretrained(model_path, use_fast=True)\n\n# tokenizer = transformers.BertTokenizer.from_pretrained(model_path)\ntokenizer.save_pretrained(\".\")","metadata":{"papermill":{"duration":0.140755,"end_time":"2022-07-20T11:17:58.325017","exception":false,"start_time":"2022-07-20T11:17:58.184262","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.480028Z","iopub.execute_input":"2022-08-03T18:45:23.480561Z","iopub.status.idle":"2022-08-03T18:45:23.595337Z","shell.execute_reply.started":"2022-08-03T18:45:23.480526Z","shell.execute_reply":"2022-08-03T18:45:23.594447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding seperator \"[SEP]\" between discourse type and discourse text ","metadata":{"papermill":{"duration":0.034397,"end_time":"2022-07-20T11:17:58.393030","exception":false,"start_time":"2022-07-20T11:17:58.358633","status":"completed"},"tags":[]}},{"cell_type":"code","source":"SEP = tokenizer.sep_token\ntrain_data['inputs'] = train_data.discourse_type + SEP + train_data.discourse_text  # noqa\ntrain_data['inputs'].iloc[0]","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.070346,"end_time":"2022-07-20T11:17:58.497112","exception":false,"start_time":"2022-07-20T11:17:58.426766","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.596943Z","iopub.execute_input":"2022-08-03T18:45:23.597324Z","iopub.status.idle":"2022-08-03T18:45:23.622738Z","shell.execute_reply.started":"2022-08-03T18:45:23.597288Z","shell.execute_reply":"2022-08-03T18:45:23.621796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Labels","metadata":{"papermill":{"duration":0.034968,"end_time":"2022-07-20T11:17:58.566744","exception":false,"start_time":"2022-07-20T11:17:58.531776","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# encode target\ntarget_encoder_dict = {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}\nnew_label = {\"discourse_effectiveness\": target_encoder_dict}\n\ntrain_data = train_data.replace(new_label)\ntrain_data = train_data.rename(columns=dict(discourse_effectiveness='label'))\ntrain_data[['label', 'inputs']].head()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.08335,"end_time":"2022-07-20T11:17:58.684536","exception":false,"start_time":"2022-07-20T11:17:58.601186","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.624466Z","iopub.execute_input":"2022-08-03T18:45:23.624968Z","iopub.status.idle":"2022-08-03T18:45:23.665563Z","shell.execute_reply.started":"2022-08-03T18:45:23.624929Z","shell.execute_reply":"2022-08-03T18:45:23.664657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split dataset","metadata":{"papermill":{"duration":0.03418,"end_time":"2022-07-20T11:17:58.754053","exception":false,"start_time":"2022-07-20T11:17:58.719873","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# split dataset\nX_train, X_valid, y_train, y_valid = train_test_split(\n    train_data['inputs'], train_data['label'],\n    test_size=0.2, random_state=42)","metadata":{"papermill":{"duration":0.049098,"end_time":"2022-07-20T11:17:58.838032","exception":false,"start_time":"2022-07-20T11:17:58.788934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.666925Z","iopub.execute_input":"2022-08-03T18:45:23.667563Z","iopub.status.idle":"2022-08-03T18:45:23.680338Z","shell.execute_reply.started":"2022-08-03T18:45:23.667527Z","shell.execute_reply":"2022-08-03T18:45:23.679392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encoder(X_train.astype(str), tokenizer)\nX_valid = bert_encoder(X_valid.astype(str), tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"papermill":{"duration":18.750019,"end_time":"2022-07-20T11:18:17.622928","exception":false,"start_time":"2022-07-20T11:17:58.872909","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:23.681902Z","iopub.execute_input":"2022-08-03T18:45:23.682440Z","iopub.status.idle":"2022-08-03T18:45:36.763243Z","shell.execute_reply.started":"2022-08-03T18:45:23.682374Z","shell.execute_reply":"2022-08-03T18:45:36.762246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## batch-wise data","metadata":{"_kg_hide-output":true,"papermill":{"duration":0.034184,"end_time":"2022-07-20T11:18:17.692045","exception":false,"start_time":"2022-07-20T11:18:17.657861","status":"completed"},"tags":[]}},{"cell_type":"code","source":"BATCH_SIZE = 8","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:36.767796Z","iopub.execute_input":"2022-08-03T18:45:36.768435Z","iopub.status.idle":"2022-08-03T18:45:36.778662Z","shell.execute_reply.started":"2022-08-03T18:45:36.768381Z","shell.execute_reply":"2022-08-03T18:45:36.777719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"_kg_hide-output":true,"papermill":{"duration":1.874168,"end_time":"2022-07-20T11:18:19.601452","exception":false,"start_time":"2022-07-20T11:18:17.727284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:36.783374Z","iopub.execute_input":"2022-08-03T18:45:36.784138Z","iopub.status.idle":"2022-08-03T18:45:36.794902Z","shell.execute_reply.started":"2022-08-03T18:45:36.784103Z","shell.execute_reply":"2022-08-03T18:45:36.793876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert dataset to tensorflow format\ntrain_dataset = tf.data.Dataset.from_tensor_slices((X_train, y_train))\n\n# Shuffle dataset to ensure even distribution of target among batches\ntrain_dataset = train_dataset.shuffle(2048)\n\n# Separate dataset in batches for training\ntrain_dataset = train_dataset.batch(BATCH_SIZE).prefetch(AUTOTUNE)\n\n# Transform dataset in \"infinite\" dataset in order to have always\n# samples of all classes in each batch\ntrain_dataset = train_dataset.repeat()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:36.799333Z","iopub.execute_input":"2022-08-03T18:45:36.799896Z","iopub.status.idle":"2022-08-03T18:45:37.207178Z","shell.execute_reply.started":"2022-08-03T18:45:36.799862Z","shell.execute_reply":"2022-08-03T18:45:37.206181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now the same for validation dataset :\nvalid_dataset = tf.data.Dataset.from_tensor_slices((X_valid, y_valid))\n\nvalid_dataset = valid_dataset.batch(BATCH_SIZE).prefetch(AUTOTUNE).cache()\nvalid_dataset = valid_dataset.repeat()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:37.208827Z","iopub.execute_input":"2022-08-03T18:45:37.209168Z","iopub.status.idle":"2022-08-03T18:45:37.288268Z","shell.execute_reply.started":"2022-08-03T18:45:37.209133Z","shell.execute_reply":"2022-08-03T18:45:37.287310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample input to BERT model ","metadata":{"papermill":{"duration":0.042091,"end_time":"2022-07-20T11:18:19.679514","exception":false,"start_time":"2022-07-20T11:18:19.637423","status":"completed"},"tags":[]}},{"cell_type":"code","source":"for k, _ in train_dataset.take(1):\n    print(k)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.373368,"end_time":"2022-07-20T11:18:20.113570","exception":false,"start_time":"2022-07-20T11:18:19.740202","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:37.293814Z","iopub.execute_input":"2022-08-03T18:45:37.294100Z","iopub.status.idle":"2022-08-03T18:45:37.504054Z","shell.execute_reply.started":"2022-08-03T18:45:37.294074Z","shell.execute_reply":"2022-08-03T18:45:37.503054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Build Model**","metadata":{"papermill":{"duration":0.056822,"end_time":"2022-07-20T11:18:20.229152","exception":false,"start_time":"2022-07-20T11:18:20.172330","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def build_model(model_bert, max_len, dropout_rate):\n    \"\"\"Build a transfer learning model based on BERT:\n    \n    * `model_bert` variable should contain pretrained model initialized by\n    TFBertModel.from_pretrained(\"/path/to/model\")  \n    * Specify the input layer size (max_len)\n    * Specify attention mask size\n    * Specify output size (3)\n    * Add a dropout layer with given `dropout_rate`\n    * Add a second dense layer to add more trainable parameters to model\n    * Compile model with optimizer Adam and learning rate\n    \"\"\"\n    input_ids = Input(shape=(max_len, ), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(\n        shape=(max_len,),\n        dtype=tf.int32,\n        name=\"token_type_ids\")\n\n    attention_mask = Input(\n        shape=(max_len,),\n        dtype=tf.int32,\n        name=\"attention_mask\")\n\n    sequence_output = model_bert.bert(\n        input_ids, token_type_ids=token_type_ids,\n        attention_mask=attention_mask)[0]\n\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(dropout_rate)(clf_output)\n\n    # ##### add another dense layer to get more trainable parameters\n    class1 = Dense(1024, activation='relu')(clf_output)\n    out = Dense(3, activation='softmax')(class1)\n    # #####\n\n    inputs = [input_ids, token_type_ids, attention_mask]\n    model = Model(inputs=inputs, outputs=out)\n    model.compile(\n        Adam(learning_rate=2e-05),\n        loss=SparseCategoricalCrossentropy(),\n        metrics=['accuracy'])\n\n    return model","metadata":{"_kg_hide-output":true,"papermill":{"duration":0.07374,"end_time":"2022-07-20T11:18:20.369698","exception":false,"start_time":"2022-07-20T11:18:20.295958","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:37.505330Z","iopub.execute_input":"2022-08-03T18:45:37.505932Z","iopub.status.idle":"2022-08-03T18:45:37.525430Z","shell.execute_reply.started":"2022-08-03T18:45:37.505893Z","shell.execute_reply":"2022-08-03T18:45:37.524318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer_layer = TFBertModel.from_pretrained(\"../input/bertbasecased/bert-base-cased\")  # noqa\nmodel = build_model(transformer_layer, max_len=256, dropout_rate=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:45:37.527649Z","iopub.execute_input":"2022-08-03T18:45:37.528060Z","iopub.status.idle":"2022-08-03T18:45:49.679749Z","shell.execute_reply.started":"2022-08-03T18:45:37.528001Z","shell.execute_reply":"2022-08-03T18:45:49.678753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_best = tf.keras.callbacks.ModelCheckpoint(\n    \"./Model.h5\",\n    monitor=\"val_accuracy\",\n    save_best_only=True,\n    verbose=1)","metadata":{"_kg_hide-output":true,"papermill":{"duration":10.066789,"end_time":"2022-07-20T11:18:30.614102","exception":false,"start_time":"2022-07-20T11:18:20.547313","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:49.681364Z","iopub.execute_input":"2022-08-03T18:45:49.682012Z","iopub.status.idle":"2022-08-03T18:45:49.691215Z","shell.execute_reply.started":"2022-08-03T18:45:49.681974Z","shell.execute_reply":"2022-08-03T18:45:49.689347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"papermill":{"duration":0.059255,"end_time":"2022-07-20T11:18:30.711660","exception":false,"start_time":"2022-07-20T11:18:30.652405","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:49.692811Z","iopub.execute_input":"2022-08-03T18:45:49.693770Z","iopub.status.idle":"2022-08-03T18:45:49.716047Z","shell.execute_reply.started":"2022-08-03T18:45:49.693718Z","shell.execute_reply":"2022-08-03T18:45:49.714970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model Training**","metadata":{"papermill":{"duration":0.037233,"end_time":"2022-07-20T11:18:30.786986","exception":false,"start_time":"2022-07-20T11:18:30.749753","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\nprint('\\n\\nModel Training..................................\\n')\nhistory = model.fit(\n    train_dataset, steps_per_epoch=350, validation_steps=350,\n    validation_data=valid_dataset, epochs=20, callbacks=[save_best])","metadata":{"papermill":{"duration":4923.474059,"end_time":"2022-07-20T12:40:34.299249","exception":false,"start_time":"2022-07-20T11:18:30.825190","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T18:45:49.717801Z","iopub.execute_input":"2022-08-03T18:45:49.718522Z","iopub.status.idle":"2022-08-03T19:30:16.834580Z","shell.execute_reply.started":"2022-08-03T18:45:49.718429Z","shell.execute_reply":"2022-08-03T19:30:16.833801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_best.best","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:30:16.836913Z","iopub.execute_input":"2022-08-03T19:30:16.837932Z","iopub.status.idle":"2022-08-03T19:30:16.846623Z","shell.execute_reply.started":"2022-08-03T19:30:16.837892Z","shell.execute_reply":"2022-08-03T19:30:16.845889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The score is lower than 0.68428 for model bert-base-cased. Adding a dense layer did not help improve accuracy","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:32:22.713134Z","iopub.execute_input":"2022-08-03T19:32:22.713545Z","iopub.status.idle":"2022-08-03T19:32:22.728961Z","shell.execute_reply.started":"2022-08-03T19:32:22.713512Z","shell.execute_reply":"2022-08-03T19:32:22.728124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['text'] = test_data.discourse_type + '[SEP]' + test_data.discourse_text  # noqa\ntest_data.head()","metadata":{"papermill":{"duration":1.881931,"end_time":"2022-07-20T12:40:38.080948","exception":false,"start_time":"2022-07-20T12:40:36.199017","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:30:16.871611Z","iopub.execute_input":"2022-08-03T19:30:16.872186Z","iopub.status.idle":"2022-08-03T19:30:16.888491Z","shell.execute_reply.started":"2022-08-03T19:30:16.872159Z","shell.execute_reply":"2022-08-03T19:30:16.887888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_text = bert_encoder(test_data.text.astype(str), tokenizer)","metadata":{"papermill":{"duration":2.093939,"end_time":"2022-07-20T12:40:42.070956","exception":false,"start_time":"2022-07-20T12:40:39.977017","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:30:16.891182Z","iopub.execute_input":"2022-08-03T19:30:16.892421Z","iopub.status.idle":"2022-08-03T19:30:16.910134Z","shell.execute_reply.started":"2022-08-03T19:30:16.892366Z","shell.execute_reply":"2022-08-03T19:30:16.909504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction**","metadata":{"papermill":{"duration":1.836399,"end_time":"2022-07-20T12:40:45.737179","exception":false,"start_time":"2022-07-20T12:40:43.900780","status":"completed"},"tags":[]}},{"cell_type":"code","source":"preds = model.predict(test_text, verbose=1)\npreds","metadata":{"papermill":{"duration":4.648509,"end_time":"2022-07-20T12:40:52.272378","exception":false,"start_time":"2022-07-20T12:40:47.623869","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:30:16.913174Z","iopub.execute_input":"2022-08-03T19:30:16.913471Z","iopub.status.idle":"2022-08-03T19:30:19.770015Z","shell.execute_reply.started":"2022-08-03T19:30:16.913445Z","shell.execute_reply":"2022-08-03T19:30:19.769456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds > 0.5","metadata":{"papermill":{"duration":1.852561,"end_time":"2022-07-20T12:40:55.984386","exception":false,"start_time":"2022-07-20T12:40:54.131825","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:30:19.773107Z","iopub.execute_input":"2022-08-03T19:30:19.773813Z","iopub.status.idle":"2022-08-03T19:30:19.782217Z","shell.execute_reply.started":"2022-08-03T19:30:19.773784Z","shell.execute_reply":"2022-08-03T19:30:19.781448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**","metadata":{"papermill":{"duration":2.213697,"end_time":"2022-07-20T12:41:00.278194","exception":false,"start_time":"2022-07-20T12:40:58.064497","status":"completed"},"tags":[]}},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:30:19.784704Z","iopub.execute_input":"2022-08-03T19:30:19.785063Z","iopub.status.idle":"2022-08-03T19:30:19.797141Z","shell.execute_reply.started":"2022-08-03T19:30:19.785028Z","shell.execute_reply":"2022-08-03T19:30:19.796149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:50:16.932066Z","iopub.execute_input":"2022-08-03T19:50:16.932461Z","iopub.status.idle":"2022-08-03T19:50:16.947144Z","shell.execute_reply.started":"2022-08-03T19:50:16.932424Z","shell.execute_reply":"2022-08-03T19:50:16.946258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data['Ineffective'] = preds[:, 0]\nsubmission_data['Adequate'] = preds[:, 1]\nsubmission_data['Effective'] = preds[:, 2]\nsubmission_data","metadata":{"papermill":{"duration":1.995519,"end_time":"2022-07-20T12:41:04.214221","exception":false,"start_time":"2022-07-20T12:41:02.218702","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:30:19.820599Z","iopub.execute_input":"2022-08-03T19:30:19.821886Z","iopub.status.idle":"2022-08-03T19:30:19.840425Z","shell.execute_reply.started":"2022-08-03T19:30:19.821838Z","shell.execute_reply":"2022-08-03T19:30:19.839761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data.to_csv(\"submission.csv\", index=False)","metadata":{"papermill":{"duration":1.861357,"end_time":"2022-07-20T12:41:08.003726","exception":false,"start_time":"2022-07-20T12:41:06.142369","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-03T19:52:01.281446Z","iopub.execute_input":"2022-08-03T19:52:01.282053Z","iopub.status.idle":"2022-08-03T19:52:01.292016Z","shell.execute_reply.started":"2022-08-03T19:52:01.282018Z","shell.execute_reply":"2022-08-03T19:52:01.291230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}