{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n<h2 id=\"adda\" style=\"color:black;background:#F4B400;padding:10px;border-radius:8px\"> Objective </h2>\n\nThe goal of this competition is to classify argumentative elements in student writing as \"effective,\" \"adequate,\" or \"ineffective.\"\n\nRubric : https://docs.google.com/document/d/1G51Ulb0i-nKCRQSs4p4ujauy4wjAJOae/edit\n\nEach essay is split into the following argumentative/discourse elements\n\n* Lead - an introduction that begins with a statistic, a quotation, a description, or some other device to grab the reader’s attention and point toward the thesis\n* Position - an opinion or conclusion on the main question\n* Claim - a claim that supports the position\n* Counterclaim - a claim that refutes another claim or gives an opposing reason to the position\n* Rebuttal - a claim that refutes a counterclaim\n* Evidence - ideas or examples that support claims, counterclaims, or rebuttals.\n* Concluding Statement - a concluding statement that restates the claims\n\nWe have to classify each of the discourse elements into the 3 categories - \"effective,\" \"adequate,\" or \"ineffective.\"","metadata":{}},{"cell_type":"markdown","source":"\n<h2 id=\"adda\" style=\"color:black;background:#F4B400;padding:10px;border-radius:8px\"> Load data and libraries </h2>\n","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:46.782847Z","iopub.execute_input":"2022-07-15T02:59:46.783487Z","iopub.status.idle":"2022-07-15T02:59:47.724272Z","shell.execute_reply.started":"2022-07-15T02:59:46.783454Z","shell.execute_reply":"2022-07-15T02:59:47.723367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer, TFBertModel\nimport matplotlib.pyplot as plt\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:47.726202Z","iopub.execute_input":"2022-07-15T02:59:47.726582Z","iopub.status.idle":"2022-07-15T02:59:54.278235Z","shell.execute_reply.started":"2022-07-15T02:59:47.726542Z","shell.execute_reply":"2022-07-15T02:59:54.277465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')\ntest = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')\nsubmission = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.279288Z","iopub.execute_input":"2022-07-15T02:59:54.279505Z","iopub.status.idle":"2022-07-15T02:59:54.610057Z","shell.execute_reply.started":"2022-07-15T02:59:54.279480Z","shell.execute_reply":"2022-07-15T02:59:54.609403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<h2 id=\"adda\" style=\"color:black;background:#F4B400;padding:10px;border-radius:8px\"> Data Exploration </h2>\n","metadata":{}},{"cell_type":"markdown","source":"1. There are 4191 essays\n2. Not all essays have all the argumentative discourse types","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.611533Z","iopub.execute_input":"2022-07-15T02:59:54.612175Z","iopub.status.idle":"2022-07-15T02:59:54.634418Z","shell.execute_reply.started":"2022-07-15T02:59:54.612138Z","shell.execute_reply":"2022-07-15T02:59:54.633647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.635503Z","iopub.execute_input":"2022-07-15T02:59:54.635839Z","iopub.status.idle":"2022-07-15T02:59:54.645124Z","shell.execute_reply.started":"2022-07-15T02:59:54.635812Z","shell.execute_reply":"2022-07-15T02:59:54.644468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.646217Z","iopub.execute_input":"2022-07-15T02:59:54.646553Z","iopub.status.idle":"2022-07-15T02:59:54.657074Z","shell.execute_reply.started":"2022-07-15T02:59:54.646526Z","shell.execute_reply":"2022-07-15T02:59:54.656489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['essay_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.657947Z","iopub.execute_input":"2022-07-15T02:59:54.658503Z","iopub.status.idle":"2022-07-15T02:59:54.678549Z","shell.execute_reply.started":"2022-07-15T02:59:54.658472Z","shell.execute_reply":"2022-07-15T02:59:54.677890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nax = sns.countplot(x='discourse_type', data=train)\nax.bar_label(ax.containers[0])\nplt.title(\"Discourse Type Distribution\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:54.679741Z","iopub.execute_input":"2022-07-15T02:59:54.680138Z","iopub.status.idle":"2022-07-15T02:59:55.142145Z","shell.execute_reply.started":"2022-07-15T02:59:54.680093Z","shell.execute_reply":"2022-07-15T02:59:55.141224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nax = sns.countplot(x='discourse_effectiveness', data=train)\nax.bar_label(ax.containers[0])\nplt.title(\"Discourse effectiveness distribution\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:55.143448Z","iopub.execute_input":"2022-07-15T02:59:55.143746Z","iopub.status.idle":"2022-07-15T02:59:55.362691Z","shell.execute_reply.started":"2022-07-15T02:59:55.143709Z","shell.execute_reply":"2022-07-15T02:59:55.361757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<h2 id=\"adda\" style=\"color:black;background:#F4B400;padding:10px;border-radius:8px\"> Modelling - BERT</h2>\n","metadata":{}},{"cell_type":"code","source":"os.environ[\"WANDB_API_KEY\"] = \"0\" ## to silence warning","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:55.366360Z","iopub.execute_input":"2022-07-15T02:59:55.366768Z","iopub.status.idle":"2022-07-15T02:59:55.371194Z","shell.execute_reply.started":"2022-07-15T02:59:55.366722Z","shell.execute_reply":"2022-07-15T02:59:55.370337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    strategy = tf.distribute.get_strategy() # for CPU and single GPU\n\nprint('Number of replicas:', strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T02:59:55.372381Z","iopub.execute_input":"2022-07-15T02:59:55.372652Z","iopub.status.idle":"2022-07-15T03:00:01.759355Z","shell.execute_reply.started":"2022-07-15T02:59:55.372603Z","shell.execute_reply":"2022-07-15T03:00:01.758377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"codes = {'Ineffective':0, 'Adequate':1, 'Effective':2}\ntrain['discourse_effectiveness'] = train['discourse_effectiveness'].map(codes)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.760811Z","iopub.execute_input":"2022-07-15T03:00:01.761178Z","iopub.status.idle":"2022-07-15T03:00:01.776430Z","shell.execute_reply.started":"2022-07-15T03:00:01.761134Z","shell.execute_reply":"2022-07-15T03:00:01.775406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer, TFBertModel\n#https://huggingface.co/transformers/model_doc/bert.html#tfbertmodel\nmodel_name = '../input/huggingface-bert-variants/bert-base-cased/bert-base-cased' #'bert-base-cased'\ntokenizer = BertTokenizer.from_pretrained(model_name)\ntokenizer.save_pretrained('.')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.778227Z","iopub.execute_input":"2022-07-15T03:00:01.778604Z","iopub.status.idle":"2022-07-15T03:00:01.888148Z","shell.execute_reply.started":"2022-07-15T03:00:01.778573Z","shell.execute_reply":"2022-07-15T03:00:01.887214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_sentence(s):\n    tokens = list(tokenizer.tokenize(s))\n    tokens.append('[SEP]')\n    return tokenizer.convert_tokens_to_ids(tokens)\n\ns = train['discourse_text'][0]\nprint(list(tokenizer.tokenize(s)))\nprint(encode_sentence(s))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.889413Z","iopub.execute_input":"2022-07-15T03:00:01.889639Z","iopub.status.idle":"2022-07-15T03:00:01.900921Z","shell.execute_reply.started":"2022-07-15T03:00:01.889612Z","shell.execute_reply":"2022-07-15T03:00:01.900248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_len = 256","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.901966Z","iopub.execute_input":"2022-07-15T03:00:01.902211Z","iopub.status.idle":"2022-07-15T03:00:01.912703Z","shell.execute_reply.started":"2022-07-15T03:00:01.902183Z","shell.execute_reply":"2022-07-15T03:00:01.911623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encode(discourse_types, discourse_texts, tokenizer):\n    num_examples = len(discourse_types)\n    discourse_types_tf = tf.ragged.constant([\n      encode_sentence(s)\n      for s in np.array(discourse_types)])\n    discourse_texts_tf = tf.ragged.constant([\n      encode_sentence(s)\n       for s in np.array(discourse_texts)])\n    \n    #input word ids\n    cls = [tokenizer.convert_tokens_to_ids(['[CLS]'])]*discourse_types_tf.shape[0]\n    input_word_ids = tf.concat([cls, discourse_types_tf, discourse_texts_tf], axis=-1)\n    \n    #input mask\n    input_mask = tf.ones_like(input_word_ids).to_tensor(shape=[num_examples, max_len])\n    \n    #input type ids\n    type_cls = tf.zeros_like(cls)\n    type_s1 = tf.zeros_like(discourse_types_tf)\n    type_s2 = tf.ones_like(discourse_texts_tf)\n    input_type_ids = tf.concat(\n    [type_cls, type_s1, type_s2], axis=-1).to_tensor(shape=[num_examples, max_len])\n    \n    inputs = {\n      'input_word_ids': input_word_ids.to_tensor(shape=[num_examples, max_len]),\n      'input_mask': input_mask,\n      'input_type_ids': input_type_ids}\n\n    return inputs","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.913902Z","iopub.execute_input":"2022-07-15T03:00:01.914210Z","iopub.status.idle":"2022-07-15T03:00:01.926019Z","shell.execute_reply.started":"2022-07-15T03:00:01.914158Z","shell.execute_reply":"2022-07-15T03:00:01.925078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_input = bert_encode(train.discourse_type.values,train.discourse_text.values, tokenizer)\ntest_input = bert_encode(test.discourse_type.values, test.discourse_text.values, tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:01.927322Z","iopub.execute_input":"2022-07-15T03:00:01.927556Z","iopub.status.idle":"2022-07-15T03:00:57.352047Z","shell.execute_reply.started":"2022-07-15T03:00:01.927522Z","shell.execute_reply":"2022-07-15T03:00:57.351143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    bert_encoder = TFBertModel.from_pretrained(model_name)\n    input_word_ids = tf.keras.Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    input_mask = tf.keras.Input(shape=(max_len,), dtype=tf.int32, name=\"input_mask\")\n    input_type_ids = tf.keras.Input(shape=(max_len,), dtype=tf.int32, name=\"input_type_ids\")\n    \n    embedding = bert_encoder([input_word_ids, input_mask, input_type_ids])[0]\n    output = tf.keras.layers.Dense(3, activation='softmax')(embedding[:,0,:])\n    \n    model = tf.keras.Model(inputs=[input_word_ids, input_mask, input_type_ids], outputs=output)\n    model.compile(tf.keras.optimizers.Adam(learning_rate=1e-5), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:00:57.353482Z","iopub.execute_input":"2022-07-15T03:00:57.353713Z","iopub.status.idle":"2022-07-15T03:00:57.362657Z","shell.execute_reply.started":"2022-07-15T03:00:57.353686Z","shell.execute_reply":"2022-07-15T03:00:57.361668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:04:36.258393Z","iopub.execute_input":"2022-07-15T03:04:36.258712Z","iopub.status.idle":"2022-07-15T03:04:51.719762Z","shell.execute_reply.started":"2022-07-15T03:04:36.258680Z","shell.execute_reply":"2022-07-15T03:04:51.718909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_input","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:03:52.307895Z","iopub.execute_input":"2022-07-15T03:03:52.308342Z","iopub.status.idle":"2022-07-15T03:03:52.602079Z","shell.execute_reply.started":"2022-07-15T03:03:52.308309Z","shell.execute_reply":"2022-07-15T03:03:52.601227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" train.discourse_effectiveness.values","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:04:23.906990Z","iopub.execute_input":"2022-07-15T03:04:23.907296Z","iopub.status.idle":"2022-07-15T03:04:23.914267Z","shell.execute_reply.started":"2022-07-15T03:04:23.907267Z","shell.execute_reply":"2022-07-15T03:04:23.913269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_input, train.discourse_effectiveness.values, epochs = 2 , verbose = 1, batch_size = 32, validation_split = 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:04:55.153271Z","iopub.execute_input":"2022-07-15T03:04:55.154033Z","iopub.status.idle":"2022-07-15T03:12:38.858472Z","shell.execute_reply.started":"2022-07-15T03:04:55.153995Z","shell.execute_reply":"2022-07-15T03:12:38.857018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_input)\nsubmission['Ineffective'] = predictions[:,0]\nsubmission['Adequate'] = predictions[:,1]\nsubmission['Effective'] = predictions[:,2]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:01:36.314721Z","iopub.status.idle":"2022-07-15T03:01:36.315646Z","shell.execute_reply.started":"2022-07-15T03:01:36.315341Z","shell.execute_reply":"2022-07-15T03:01:36.315371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:01:36.317142Z","iopub.status.idle":"2022-07-15T03:01:36.317969Z","shell.execute_reply.started":"2022-07-15T03:01:36.317691Z","shell.execute_reply":"2022-07-15T03:01:36.317727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:01:36.319031Z","iopub.status.idle":"2022-07-15T03:01:36.319433Z","shell.execute_reply.started":"2022-07-15T03:01:36.319232Z","shell.execute_reply":"2022-07-15T03:01:36.319265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}