{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Abstract\n\nTrain and prototype your models quickly by using TPUs. This notebook shows easy and quick way to inference 🤗Transformers on TPUs.","metadata":{}},{"cell_type":"markdown","source":"# 📝 Versions\n[RoBerta base TPU Training Notebook](https://www.kaggle.com/code/bharadwajvedula/feedback-prize-tpu-roberta-training-fold-4/notebook)\n\nVersion 2: Roberta Large **CV: 0.836 LB: 0.797**\n\nVersion 3: Roberta Large with self change in dataset **CV:- 0.704 LB:- 0.**","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\nfrom text_unidecode import unidecode\nfrom typing import Dict, List, Tuple\nimport codecs\n\nfrom statistics import mode\nfrom scipy.stats import pearsonr\n\nfrom transformers import TFAutoModel, AutoTokenizer\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy\nfrom tensorflow.keras.activations import tanh, softmax\nfrom tensorflow.keras.layers import Layer,Input, Dense, Flatten, Dropout, GlobalAveragePooling1D\nfrom tensorflow.keras.models import Model, save_model, load_model\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau , EarlyStopping\nfrom tensorflow.keras.optimizers import Adam, SGD\n\nimport os\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:37:30.059798Z","iopub.execute_input":"2022-07-20T23:37:30.060219Z","iopub.status.idle":"2022-07-20T23:37:36.215561Z","shell.execute_reply.started":"2022-07-20T23:37:30.060130Z","shell.execute_reply":"2022-07-20T23:37:36.214379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Config","metadata":{}},{"cell_type":"code","source":"class config:\n    base_dir = \"../input/feedback-prize-effectiveness/\"\n    # dataset path \n    train_dataset_path = \"../input/feedbackprizegroupkfolds/train.csv\"\n    test_dataset_path = \"../input/feedback-prize-effectiveness/test.csv\"\n    sample_submission_path = \"../input/feedback-prize-effectiveness/sample_submission.csv\"\n       \n    save_dir=\"./result\"\n    \n    AUTOTUNE = tf.data.AUTOTUNE\n    \n    #tokenizer params\n    truncation = True \n    padding = 'max_length'\n    max_length = 512\n    tokenizer_path = \"../input/transformers/roberta-large\"\n    \n    # model params\n    model_name = \"roberta-large\"\n    hf_model = \"../input/feedbackprizerobertalarge-fold4/result/roberta-large\"\n    model_weights = \"../input/feedbackprizerobertalarge-fold123/result\"\n    model_weights_2 = \"../input/feedbackprizerobertalarge-fold4/result\"\n\n    \n    #training params\n    learning_rate = 1e-5\n    batch_size = 24\n    epochs = 12\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T23:38:49.629921Z","iopub.execute_input":"2022-07-20T23:38:49.630594Z","iopub.status.idle":"2022-07-20T23:38:49.639306Z","shell.execute_reply.started":"2022-07-20T23:38:49.630551Z","shell.execute_reply":"2022-07-20T23:38:49.638171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📊 Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_test_essay(essay_id):\n    parent_path = config.base_dir + 'test'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:38:10.138426Z","iopub.execute_input":"2022-07-20T23:38:10.139071Z","iopub.status.idle":"2022-07-20T23:38:10.144829Z","shell.execute_reply.started":"2022-07-20T23:38:10.139034Z","shell.execute_reply":"2022-07-20T23:38:10.143396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:38:20.489991Z","iopub.execute_input":"2022-07-20T23:38:20.490891Z","iopub.status.idle":"2022-07-20T23:38:20.501095Z","shell.execute_reply.started":"2022-07-20T23:38:20.490841Z","shell.execute_reply":"2022-07-20T23:38:20.500112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(config.train_dataset_path)\ndf_test = pd.read_csv(config.test_dataset_path)\ndf_ss = pd.read_csv(config.sample_submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:38:30.429244Z","iopub.execute_input":"2022-07-20T23:38:30.429618Z","iopub.status.idle":"2022-07-20T23:38:30.728638Z","shell.execute_reply.started":"2022-07-20T23:38:30.429583Z","shell.execute_reply":"2022-07-20T23:38:30.727656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['essay_text'] = df_test['essay_id'].apply(get_test_essay)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:38:52.808604Z","iopub.execute_input":"2022-07-20T23:38:52.808938Z","iopub.status.idle":"2022-07-20T23:38:52.824852Z","shell.execute_reply.started":"2022-07-20T23:38:52.808908Z","shell.execute_reply":"2022-07-20T23:38:52.823950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['discourse_text'] = df_test['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_test['essay_text'] = df_test['essay_text'].apply(resolve_encodings_and_normalize)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:39:35.996992Z","iopub.execute_input":"2022-07-20T23:39:35.997681Z","iopub.status.idle":"2022-07-20T23:39:36.011045Z","shell.execute_reply.started":"2022-07-20T23:39:35.997642Z","shell.execute_reply":"2022-07-20T23:39:36.010040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['text'] = df_test['discourse_type'] + \" [SEP] \" + df_test['discourse_text'] + \" [SEP] \" + df_test['essay_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:39:36.028228Z","iopub.execute_input":"2022-07-20T23:39:36.028495Z","iopub.status.idle":"2022-07-20T23:39:36.037015Z","shell.execute_reply.started":"2022-07-20T23:39:36.028470Z","shell.execute_reply":"2022-07-20T23:39:36.034609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🎟 Tokenizer","metadata":{}},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.tokenizer_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:05.678447Z","iopub.execute_input":"2022-07-20T23:40:05.678789Z","iopub.status.idle":"2022-07-20T23:40:05.855868Z","shell.execute_reply.started":"2022-07-20T23:40:05.678757Z","shell.execute_reply":"2022-07-20T23:40:05.854785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = tokenizer(\n        df_test['text'].tolist(),\n        truncation = config.truncation, \n        padding = config.padding,\n        max_length =config.max_length   \n    )","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:10.279097Z","iopub.execute_input":"2022-07-20T23:40:10.280154Z","iopub.status.idle":"2022-07-20T23:40:10.308232Z","shell.execute_reply.started":"2022-07-20T23:40:10.280107Z","shell.execute_reply":"2022-07-20T23:40:10.307009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧰 Dataset Prep Function","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef map_function(encodings):\n    input_ids = encodings['input_ids']\n    attention_mask = encodings['attention_mask']\n        \n    return {'input_ids': input_ids , 'attention_mask': attention_mask}","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:14.548818Z","iopub.execute_input":"2022-07-20T23:40:14.549491Z","iopub.status.idle":"2022-07-20T23:40:14.555506Z","shell.execute_reply.started":"2022-07-20T23:40:14.549453Z","shell.execute_reply":"2022-07-20T23:40:14.554186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = tf.data.Dataset.from_tensor_slices((test_embeddings))\ntest = (\n                test\n                .map(map_function, num_parallel_calls= config.AUTOTUNE)\n                .batch(config.batch_size)\n                .prefetch(config.AUTOTUNE)\n            )","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:17.808833Z","iopub.execute_input":"2022-07-20T23:40:17.809688Z","iopub.status.idle":"2022-07-20T23:40:20.546357Z","shell.execute_reply.started":"2022-07-20T23:40:17.809648Z","shell.execute_reply":"2022-07-20T23:40:20.545378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧠 Model","metadata":{}},{"cell_type":"code","source":"def create_model():\n    input_id = Input(shape = (config.max_length) , dtype = tf.int32, name = 'input_ids')\n    attention_mask = Input(shape = (config.max_length), dtype = tf.int32, name = 'attention_mask')\n    \n    transformer_model = TFAutoModel.from_pretrained(config.hf_model)\n    cls_token = transformer_model(input_ids = input_id , attention_mask = attention_mask)[0][:,0,:]\n    \n    prediction = Dense(3 , activation = \"softmax\")(cls_token)\n\n    return Model(inputs = [input_id, attention_mask] , outputs = prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:29.590940Z","iopub.execute_input":"2022-07-20T23:40:29.591337Z","iopub.status.idle":"2022-07-20T23:40:29.599065Z","shell.execute_reply.started":"2022-07-20T23:40:29.591304Z","shell.execute_reply":"2022-07-20T23:40:29.597131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:40:32.286454Z","iopub.execute_input":"2022-07-20T23:40:32.286807Z","iopub.status.idle":"2022-07-20T23:40:58.076717Z","shell.execute_reply.started":"2022-07-20T23:40:32.286775Z","shell.execute_reply":"2022-07-20T23:40:58.075765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔄 KFold Prediction","metadata":{}},{"cell_type":"code","source":"model_paths = sorted(glob(f'{config.model_weights}/roberta-large_*'))\nmodel_paths.extend(sorted(glob(f'{config.model_weights_2}/roberta-large_*')))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:41:03.869056Z","iopub.execute_input":"2022-07-20T23:41:03.870237Z","iopub.status.idle":"2022-07-20T23:41:03.884359Z","shell.execute_reply.started":"2022-07-20T23:41:03.870198Z","shell.execute_reply":"2022-07-20T23:41:03.883321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\nfor fold in range(1,5):\n    print(f\"====== FOLD RUNNING {fold}======\") \n    \n    #Clearing backend session\n    K.clear_session()\n    print(\"Backend Cleared\")\n    \n    #loading weights\n    print(\"loading model\")\n    model.load_weights(model_paths[fold-1])\n    \n    #predictions\n    pred = model.predict(test , verbose = 1)\n    predictions.append(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:41:32.629364Z","iopub.execute_input":"2022-07-20T23:41:32.629717Z","iopub.status.idle":"2022-07-20T23:42:24.540144Z","shell.execute_reply.started":"2022-07-20T23:41:32.629685Z","shell.execute_reply":"2022-07-20T23:42:24.539116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Post Processing","metadata":{}},{"cell_type":"code","source":"final_predictions = np.mean([predictions[0],predictions[1],predictions[2],predictions[3]] , axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:42:40.097472Z","iopub.execute_input":"2022-07-20T23:42:40.098118Z","iopub.status.idle":"2022-07-20T23:42:40.109630Z","shell.execute_reply.started":"2022-07-20T23:42:40.098065Z","shell.execute_reply":"2022-07-20T23:42:40.108033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adequate = []\neffective = []\nineffective = []\n\nfor prediction in final_predictions:\n    adequate.append(prediction[0])\n    effective.append(prediction[1])\n    ineffective.append(prediction[2])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:42:48.248770Z","iopub.execute_input":"2022-07-20T23:42:48.249176Z","iopub.status.idle":"2022-07-20T23:42:48.254885Z","shell.execute_reply.started":"2022-07-20T23:42:48.249144Z","shell.execute_reply":"2022-07-20T23:42:48.253898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💯Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'discourse_id':df_test['discourse_id'],'Adequate':adequate,'Effective':effective,'Ineffective':ineffective})\nsubmission.to_csv(\"submission.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:42:51.649260Z","iopub.execute_input":"2022-07-20T23:42:51.649628Z","iopub.status.idle":"2022-07-20T23:42:51.661888Z","shell.execute_reply.started":"2022-07-20T23:42:51.649595Z","shell.execute_reply":"2022-07-20T23:42:51.660699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T23:42:52.688708Z","iopub.execute_input":"2022-07-20T23:42:52.689079Z","iopub.status.idle":"2022-07-20T23:42:52.707865Z","shell.execute_reply.started":"2022-07-20T23:42:52.689048Z","shell.execute_reply":"2022-07-20T23:42:52.706796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}