{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-20T19:24:23.190366Z","iopub.execute_input":"2022-07-20T19:24:23.190659Z","iopub.status.idle":"2022-07-20T19:24:25.385264Z","shell.execute_reply.started":"2022-07-20T19:24:23.190631Z","shell.execute_reply":"2022-07-20T19:24:25.380998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nimport transformers\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel\nfrom transformers import AutoTokenizer\n\n\ntf.config.experimental_run_functions_eagerly(False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:24:25.387370Z","iopub.execute_input":"2022-07-20T19:24:25.387705Z","iopub.status.idle":"2022-07-20T19:24:34.353408Z","shell.execute_reply.started":"2022-07-20T19:24:25.387661Z","shell.execute_reply":"2022-07-20T19:24:34.352422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:24:34.354682Z","iopub.execute_input":"2022-07-20T19:24:34.354944Z","iopub.status.idle":"2022-07-20T19:24:34.715932Z","shell.execute_reply.started":"2022-07-20T19:24:34.354916Z","shell.execute_reply":"2022-07-20T19:24:34.715159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encoder(texts,tokenizer,max_len = 256):\n    input_ids  = list()\n    token_type_ids = list()\n    attention_mask = list()\n    \n    for text in texts:\n        token  = tokenizer(text,max_length = 256,truncation = True,padding = 'max_length',add_special_tokens = True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n        \n    return np.array(input_ids),np.array(token_type_ids),np.array(attention_mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:24:34.717129Z","iopub.execute_input":"2022-07-20T19:24:34.717423Z","iopub.status.idle":"2022-07-20T19:24:34.725232Z","shell.execute_reply.started":"2022-07-20T19:24:34.717382Z","shell.execute_reply":"2022-07-20T19:24:34.724218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bert Tokenizer\nmodel_path = '../input/huggingface-bert-variants/bert-base-cased/bert-base-cased'\ntokenizer = AutoTokenizer.from_pretrained(model_path, use_fast=True)\n\n# tokenizer = transformers.BertTokenizer.from_pretrained(model_path)\ntokenizer.save_pretrained(\".\")","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:24:34.727667Z","iopub.execute_input":"2022-07-20T19:24:34.728182Z","iopub.status.idle":"2022-07-20T19:24:34.863339Z","shell.execute_reply.started":"2022-07-20T19:24:34.728138Z","shell.execute_reply":"2022-07-20T19:24:34.862308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEP = tokenizer.sep_token\ndata['inputs'] = data.discourse_type + SEP + data.discourse_text","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:24:34.864854Z","iopub.execute_input":"2022-07-20T19:24:34.865284Z","iopub.status.idle":"2022-07-20T19:24:34.904326Z","shell.execute_reply.started":"2022-07-20T19:24:34.865242Z","shell.execute_reply":"2022-07-20T19:24:34.903316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['inputs'].iloc[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:39.465772Z","iopub.execute_input":"2022-07-20T19:25:39.466268Z","iopub.status.idle":"2022-07-20T19:25:39.471610Z","shell.execute_reply.started":"2022-07-20T19:25:39.466232Z","shell.execute_reply":"2022-07-20T19:25:39.471013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating label\n\nnew_label = {\"discourse_effectiveness\": {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}}\ndata = data.replace(new_label)\ndata = data.rename(columns={\"discourse_effectiveness\":'label'})","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:39.679635Z","iopub.execute_input":"2022-07-20T19:25:39.680219Z","iopub.status.idle":"2022-07-20T19:25:39.733885Z","shell.execute_reply.started":"2022-07-20T19:25:39.680170Z","shell.execute_reply":"2022-07-20T19:25:39.733261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:39.735395Z","iopub.execute_input":"2022-07-20T19:25:39.735821Z","iopub.status.idle":"2022-07-20T19:25:39.748035Z","shell.execute_reply.started":"2022-07-20T19:25:39.735792Z","shell.execute_reply":"2022-07-20T19:25:39.747118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:40.081764Z","iopub.execute_input":"2022-07-20T19:25:40.082088Z","iopub.status.idle":"2022-07-20T19:25:46.268381Z","shell.execute_reply.started":"2022-07-20T19:25:40.082053Z","shell.execute_reply":"2022-07-20T19:25:46.267596Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split dataset\nfrom sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(data['inputs'], data['label'], test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:46.269790Z","iopub.execute_input":"2022-07-20T19:25:46.270522Z","iopub.status.idle":"2022-07-20T19:25:47.040254Z","shell.execute_reply.started":"2022-07-20T19:25:46.270489Z","shell.execute_reply":"2022-07-20T19:25:47.039543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encoder(X_train.astype(str),tokenizer)\nX_valid = bert_encoder(X_valid.astype(str),tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:25:47.041215Z","iopub.execute_input":"2022-07-20T19:25:47.041942Z","iopub.status.idle":"2022-07-20T19:26:09.474537Z","shell.execute_reply.started":"2022-07-20T19:25:47.041905Z","shell.execute_reply":"2022-07-20T19:26:09.473587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\n# batch_size = 8,lr = 3e-6,max_len = 512,\ntrain_dataset = (tf.data.Dataset.from_tensor_slices((X_train,y_train)).repeat().shuffle(2048).batch(16).prefetch(AUTOTUNE))\nvalid_dataset = (tf.data.Dataset.from_tensor_slices((X_valid,y_valid)).batch(16).cache().prefetch(AUTOTUNE))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:26:09.476952Z","iopub.execute_input":"2022-07-20T19:26:09.477334Z","iopub.status.idle":"2022-07-20T19:26:10.139949Z","shell.execute_reply.started":"2022-07-20T19:26:09.477289Z","shell.execute_reply":"2022-07-20T19:26:10.139249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:26:10.141428Z","iopub.execute_input":"2022-07-20T19:26:10.141914Z","iopub.status.idle":"2022-07-20T19:26:10.149376Z","shell.execute_reply.started":"2022-07-20T19:26:10.141882Z","shell.execute_reply":"2022-07-20T19:26:10.148110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(model_bert,max_len = 256):\n    input_ids =      Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"token_type_ids\")\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\n    sequence_output = model_bert.bert(input_ids, token_type_ids=token_type_ids, attention_mask=attention_mask)[0]\n\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(0.1)(clf_output)\n    out = Dense(3,activation = 'softmax')(clf_output)\n\n    model = Model(inputs = [input_ids,token_type_ids,attention_mask],outputs = out)\n    model.compile(Adam(learning_rate = 1e-05),loss = 'sparse_categorical_crossentropy',metrics = ['accuracy'])\n    \n    return model\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:26:10.150829Z","iopub.execute_input":"2022-07-20T19:26:10.151407Z","iopub.status.idle":"2022-07-20T19:26:10.160410Z","shell.execute_reply.started":"2022-07-20T19:26:10.151372Z","shell.execute_reply":"2022-07-20T19:26:10.159580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer_layer = (TFBertModel.from_pretrained('../input/huggingface-bert-variants/bert-base-cased/bert-base-cased'))\nmodel = build_model(transformer_layer, max_len=256)\n\nsave_best = tf.keras.callbacks.ModelCheckpoint(\"./Model.h5\", monitor='val_accuracy',save_best_only=True, verbose=1)\n\nmodel.summary()\nprint('\\n\\nModel Training..........................................\\n')\nmodel.fit(train_dataset,steps_per_epoch=350, validation_data=valid_dataset, epochs=1, callbacks=[save_best])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:27:11.748329Z","iopub.execute_input":"2022-07-20T19:27:11.748810Z","iopub.status.idle":"2022-07-20T19:31:34.025007Z","shell.execute_reply.started":"2022-07-20T19:27:11.748775Z","shell.execute_reply":"2022-07-20T19:31:34.023571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:33:26.193960Z","iopub.execute_input":"2022-07-20T19:33:26.194892Z","iopub.status.idle":"2022-07-20T19:33:27.568328Z","shell.execute_reply.started":"2022-07-20T19:33:26.194841Z","shell.execute_reply":"2022-07-20T19:33:27.567116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model('./Model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:33:27.587573Z","iopub.execute_input":"2022-07-20T19:33:27.587912Z","iopub.status.idle":"2022-07-20T19:33:33.355900Z","shell.execute_reply.started":"2022-07-20T19:33:27.587878Z","shell.execute_reply":"2022-07-20T19:33:33.354855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ntest['text'] = test.discourse_type + '[SEP]' +test.discourse_text\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:33:33.359017Z","iopub.execute_input":"2022-07-20T19:33:33.359290Z","iopub.status.idle":"2022-07-20T19:33:33.384937Z","shell.execute_reply.started":"2022-07-20T19:33:33.359261Z","shell.execute_reply":"2022-07-20T19:33:33.384011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_text = bert_encoder(test.text.astype(str), tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:53:28.394684Z","iopub.execute_input":"2022-07-19T20:53:28.395034Z","iopub.status.idle":"2022-07-19T20:53:28.407327Z","shell.execute_reply.started":"2022-07-19T20:53:28.395000Z","shell.execute_reply":"2022-07-19T20:53:28.406436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\nsubmission_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:53:28.760583Z","iopub.execute_input":"2022-07-19T20:53:28.761071Z","iopub.status.idle":"2022-07-19T20:53:28.777153Z","shell.execute_reply.started":"2022-07-19T20:53:28.761028Z","shell.execute_reply":"2022-07-19T20:53:28.776423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_text, verbose=1)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:53:29.313351Z","iopub.execute_input":"2022-07-19T20:53:29.314340Z","iopub.status.idle":"2022-07-19T20:53:32.864663Z","shell.execute_reply.started":"2022-07-19T20:53:29.314296Z","shell.execute_reply":"2022-07-19T20:53:32.863838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds > 0.5","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:53:32.866302Z","iopub.execute_input":"2022-07-19T20:53:32.866533Z","iopub.status.idle":"2022-07-19T20:53:32.872260Z","shell.execute_reply.started":"2022-07-19T20:53:32.866507Z","shell.execute_reply":"2022-07-19T20:53:32.871399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data['Ineffective'] = preds[:,0]\nsubmission_data['Adequate'] = preds[:,1]\nsubmission_data['Effective'] = preds[:,2]\nsubmission_data","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:53:32.873578Z","iopub.execute_input":"2022-07-19T20:53:32.873943Z","iopub.status.idle":"2022-07-19T20:53:32.894902Z","shell.execute_reply.started":"2022-07-19T20:53:32.873914Z","shell.execute_reply":"2022-07-19T20:53:32.893957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:18:27.530380Z","iopub.execute_input":"2022-07-19T20:18:27.531190Z","iopub.status.idle":"2022-07-19T20:18:27.542702Z","shell.execute_reply.started":"2022-07-19T20:18:27.531053Z","shell.execute_reply":"2022-07-19T20:18:27.541558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T20:18:27.544208Z","iopub.execute_input":"2022-07-19T20:18:27.544601Z","iopub.status.idle":"2022-07-19T20:18:27.567467Z","shell.execute_reply.started":"2022-07-19T20:18:27.544563Z","shell.execute_reply":"2022-07-19T20:18:27.566711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}