{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Abstract\n\nTrain and prototype your models quickly by using TPUs. This notebook shows easy and quick way to inference 🤗Transformers on TPUs.","metadata":{}},{"cell_type":"markdown","source":"# 📝 Versions\n[RoBerta Large TPU Training Notebook](https://www.kaggle.com/code/bharadwajvedula/feedback-prize-tpu-roberta-training-fold-4/notebook)\n\nVersion 2: Roberta Large **CV: 0.836 LB: 0.797**\n\nVersion 3: Roberta Large with self change in dataset **CV:- 0.704 LB:- 0.671**","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\nfrom text_unidecode import unidecode\nfrom typing import Dict, List, Tuple\nimport codecs\n\nfrom statistics import mode\nfrom scipy.stats import pearsonr\n\nfrom transformers import TFAutoModel, AutoTokenizer\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy\nfrom tensorflow.keras.activations import tanh, softmax\nfrom tensorflow.keras.layers import Layer,Input, Dense, Flatten, Dropout, GlobalAveragePooling1D\nfrom tensorflow.keras.models import Model, save_model, load_model\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau , EarlyStopping\nfrom tensorflow.keras.optimizers import Adam, SGD\n\nimport os\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:47.171423Z","iopub.execute_input":"2022-07-21T18:00:47.171849Z","iopub.status.idle":"2022-07-21T18:00:53.268921Z","shell.execute_reply.started":"2022-07-21T18:00:47.171738Z","shell.execute_reply":"2022-07-21T18:00:53.267946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Config","metadata":{}},{"cell_type":"code","source":"class config:\n    base_dir = \"../input/feedback-prize-effectiveness/\"\n    # dataset path \n    train_dataset_path = \"../input/feedbackprizegroupkfolds/train.csv\"\n    test_dataset_path = \"../input/feedback-prize-effectiveness/test.csv\"\n    sample_submission_path = \"../input/feedback-prize-effectiveness/sample_submission.csv\"\n       \n    save_dir=\"./result\"\n    \n    AUTOTUNE = tf.data.AUTOTUNE\n    \n    #tokenizer params\n    truncation = True \n    padding = 'max_length'\n    max_length = 512\n    tokenizer_path = \"../input/transformers/roberta-large\"\n    \n    # model params\n    model_name = \"roberta-large\"\n    hf_model = \"../input/feedbackprizerobertalarge-fold4/result/roberta-large\"\n    model_weights = \"../input/feedbackprizerobertalarge-fold123/result\"\n    model_weights_2 = \"../input/feedbackprizerobertalarge-fold4/result\"\n\n    \n    #training params\n    learning_rate = 1e-5\n    batch_size = 24\n    epochs = 12\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T18:00:53.270968Z","iopub.execute_input":"2022-07-21T18:00:53.271587Z","iopub.status.idle":"2022-07-21T18:00:53.279843Z","shell.execute_reply.started":"2022-07-21T18:00:53.271549Z","shell.execute_reply":"2022-07-21T18:00:53.276850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📊 Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_test_essay(essay_id):\n    parent_path = config.base_dir + 'test'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.281079Z","iopub.execute_input":"2022-07-21T18:00:53.281427Z","iopub.status.idle":"2022-07-21T18:00:53.296379Z","shell.execute_reply.started":"2022-07-21T18:00:53.281388Z","shell.execute_reply":"2022-07-21T18:00:53.295352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.299290Z","iopub.execute_input":"2022-07-21T18:00:53.299639Z","iopub.status.idle":"2022-07-21T18:00:53.308374Z","shell.execute_reply.started":"2022-07-21T18:00:53.299595Z","shell.execute_reply":"2022-07-21T18:00:53.307274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(config.train_dataset_path)\ndf_test = pd.read_csv(config.test_dataset_path)\ndf_ss = pd.read_csv(config.sample_submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.309756Z","iopub.execute_input":"2022-07-21T18:00:53.310322Z","iopub.status.idle":"2022-07-21T18:00:53.610114Z","shell.execute_reply.started":"2022-07-21T18:00:53.310286Z","shell.execute_reply":"2022-07-21T18:00:53.609158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['essay_text'] = df_test['essay_id'].apply(get_test_essay)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.611564Z","iopub.execute_input":"2022-07-21T18:00:53.611923Z","iopub.status.idle":"2022-07-21T18:00:53.629859Z","shell.execute_reply.started":"2022-07-21T18:00:53.611887Z","shell.execute_reply":"2022-07-21T18:00:53.628673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['discourse_text'] = df_test['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_test['essay_text'] = df_test['essay_text'].apply(resolve_encodings_and_normalize)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.631535Z","iopub.execute_input":"2022-07-21T18:00:53.631913Z","iopub.status.idle":"2022-07-21T18:00:53.647071Z","shell.execute_reply.started":"2022-07-21T18:00:53.631876Z","shell.execute_reply":"2022-07-21T18:00:53.646223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['text'] = df_test['discourse_type'] + \" [SEP] \" + df_test['discourse_text'] + \" [SEP] \" + df_test['essay_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.648270Z","iopub.execute_input":"2022-07-21T18:00:53.648584Z","iopub.status.idle":"2022-07-21T18:00:53.656757Z","shell.execute_reply.started":"2022-07-21T18:00:53.648551Z","shell.execute_reply":"2022-07-21T18:00:53.655785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🎟 Tokenizer","metadata":{}},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.tokenizer_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.658639Z","iopub.execute_input":"2022-07-21T18:00:53.659224Z","iopub.status.idle":"2022-07-21T18:00:53.824179Z","shell.execute_reply.started":"2022-07-21T18:00:53.659186Z","shell.execute_reply":"2022-07-21T18:00:53.823208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = tokenizer(\n        df_test['text'].tolist(),\n        truncation = config.truncation, \n        padding = config.padding,\n        max_length =config.max_length   \n    )","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.828245Z","iopub.execute_input":"2022-07-21T18:00:53.828512Z","iopub.status.idle":"2022-07-21T18:00:53.854095Z","shell.execute_reply.started":"2022-07-21T18:00:53.828487Z","shell.execute_reply":"2022-07-21T18:00:53.853252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧰 Dataset Prep Function","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef map_function(encodings):\n    input_ids = encodings['input_ids']\n    attention_mask = encodings['attention_mask']\n        \n    return {'input_ids': input_ids , 'attention_mask': attention_mask}","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.855511Z","iopub.execute_input":"2022-07-21T18:00:53.855925Z","iopub.status.idle":"2022-07-21T18:00:53.864039Z","shell.execute_reply.started":"2022-07-21T18:00:53.855887Z","shell.execute_reply":"2022-07-21T18:00:53.863166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = tf.data.Dataset.from_tensor_slices((test_embeddings))\ntest = (\n                test\n                .map(map_function, num_parallel_calls= config.AUTOTUNE)\n                .batch(config.batch_size)\n                .prefetch(config.AUTOTUNE)\n            )","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:53.867612Z","iopub.execute_input":"2022-07-21T18:00:53.868254Z","iopub.status.idle":"2022-07-21T18:00:56.771666Z","shell.execute_reply.started":"2022-07-21T18:00:53.868218Z","shell.execute_reply":"2022-07-21T18:00:56.770734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧠 Model","metadata":{}},{"cell_type":"code","source":"def create_model():\n    input_id = Input(shape = (config.max_length) , dtype = tf.int32, name = 'input_ids')\n    attention_mask = Input(shape = (config.max_length), dtype = tf.int32, name = 'attention_mask')\n    \n    transformer_model = TFAutoModel.from_pretrained(config.hf_model)\n    cls_token = transformer_model(input_ids = input_id , attention_mask = attention_mask)[0][:,0,:]\n    \n    prediction = Dense(3 , activation = \"softmax\")(cls_token)\n\n    return Model(inputs = [input_id, attention_mask] , outputs = prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:56.773255Z","iopub.execute_input":"2022-07-21T18:00:56.773931Z","iopub.status.idle":"2022-07-21T18:00:56.781869Z","shell.execute_reply.started":"2022-07-21T18:00:56.773893Z","shell.execute_reply":"2022-07-21T18:00:56.780724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:00:56.784875Z","iopub.execute_input":"2022-07-21T18:00:56.785528Z","iopub.status.idle":"2022-07-21T18:01:26.006841Z","shell.execute_reply.started":"2022-07-21T18:00:56.785491Z","shell.execute_reply":"2022-07-21T18:01:26.005937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔄 KFold Prediction","metadata":{}},{"cell_type":"code","source":"model_paths = sorted(glob(f'{config.model_weights}/roberta-large_*'))\nmodel_paths.extend(sorted(glob(f'{config.model_weights_2}/roberta-large_*')))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:01:26.008327Z","iopub.execute_input":"2022-07-21T18:01:26.008905Z","iopub.status.idle":"2022-07-21T18:01:26.022770Z","shell.execute_reply.started":"2022-07-21T18:01:26.008867Z","shell.execute_reply":"2022-07-21T18:01:26.021785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\nfor fold in range(1,5):\n    print(f\"====== FOLD RUNNING {fold}======\") \n    \n    #Clearing backend session\n    K.clear_session()\n    print(\"Backend Cleared\")\n    \n    #loading weights\n    print(\"loading model\")\n    model.load_weights(model_paths[fold-1])\n    \n    #predictions\n    pred = model.predict(test , verbose = 1)\n    predictions.append(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:01:26.028434Z","iopub.execute_input":"2022-07-21T18:01:26.029024Z","iopub.status.idle":"2022-07-21T18:02:25.130183Z","shell.execute_reply.started":"2022-07-21T18:01:26.028991Z","shell.execute_reply":"2022-07-21T18:02:25.129283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Post Processing","metadata":{}},{"cell_type":"code","source":"final_predictions = np.mean([predictions[0],predictions[1],predictions[2],predictions[3]] , axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:02:25.131726Z","iopub.execute_input":"2022-07-21T18:02:25.132368Z","iopub.status.idle":"2022-07-21T18:02:25.138001Z","shell.execute_reply.started":"2022-07-21T18:02:25.132324Z","shell.execute_reply":"2022-07-21T18:02:25.136926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adequate = []\neffective = []\nineffective = []\n\nfor prediction in final_predictions:\n    adequate.append(prediction[0])\n    effective.append(prediction[1])\n    ineffective.append(prediction[2])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:02:25.140009Z","iopub.execute_input":"2022-07-21T18:02:25.140848Z","iopub.status.idle":"2022-07-21T18:02:25.147118Z","shell.execute_reply.started":"2022-07-21T18:02:25.140794Z","shell.execute_reply":"2022-07-21T18:02:25.146186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💯Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'discourse_id':df_test['discourse_id'],'Adequate':adequate,'Effective':effective,'Ineffective':ineffective})\nsubmission.to_csv(\"submission.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:02:25.148652Z","iopub.execute_input":"2022-07-21T18:02:25.149086Z","iopub.status.idle":"2022-07-21T18:02:25.161762Z","shell.execute_reply.started":"2022-07-21T18:02:25.149050Z","shell.execute_reply":"2022-07-21T18:02:25.160829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:02:25.163427Z","iopub.execute_input":"2022-07-21T18:02:25.163998Z","iopub.status.idle":"2022-07-21T18:02:25.181992Z","shell.execute_reply.started":"2022-07-21T18:02:25.163962Z","shell.execute_reply":"2022-07-21T18:02:25.181163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}