{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 📝 Versions\n\nVersion 5: Roberta Large fold 4 **CV:- 0.836 LB:- 0.797**\n\nVersion 7: Roberta Large with self change in dataset **CV:- 0.704 LB:- 0.**\n","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Imports","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np \nimport pandas as pd \n\nfrom text_unidecode import unidecode\nfrom typing import Dict, List, Tuple\nimport codecs\n\nfrom sklearn.metrics import log_loss\n\nfrom transformers import TFAutoModel, AutoTokenizer\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy \nfrom tensorflow.keras.activations import tanh, softmax\nfrom tensorflow.keras.layers import Layer,Input, Dense, Flatten, Dropout, GlobalAveragePooling1D\nfrom tensorflow.keras.models import Model, save_model, load_model\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau , EarlyStopping\nfrom tensorflow.keras.optimizers import Adam, SGD\n\nif not os.path.exists(\"./result\"):\n    os.makedirs(\"./result\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T22:38:27.197915Z","iopub.execute_input":"2022-07-20T22:38:27.198193Z","iopub.status.idle":"2022-07-20T22:38:27.220098Z","shell.execute_reply.started":"2022-07-20T22:38:27.198167Z","shell.execute_reply":"2022-07-20T22:38:27.219086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Config","metadata":{}},{"cell_type":"code","source":"class config:\n    base_dir = \"../input/feedback-prize-effectiveness/\"\n    # dataset path \n    train_dataset_path = \"../input/feedbackprizegroupkfolds/train.csv\"\n    test_dataset_path = \"../input/feedbackprizegroupkfolds/test.csv\"\n    sample_submission_path = \"../input/feedbackprizegroupkfolds/sample_submission.csv\"\n       \n    save_dir=\"./result\"\n    \n    AUTOTUNE = tf.data.AUTOTUNE\n    \n    #tokenizer params\n    truncation = True \n    padding = 'max_length'\n    max_length = 512\n    \n    # model params\n    model_name = \"roberta-large\"\n    \n    #training params\n    learning_rate = 1e-5\n    batch_size = 24\n    epochs = 12\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:41:43.378224Z","iopub.execute_input":"2022-07-20T22:41:43.378795Z","iopub.status.idle":"2022-07-20T22:41:43.386569Z","shell.execute_reply.started":"2022-07-20T22:41:43.378728Z","shell.execute_reply":"2022-07-20T22:41:43.385656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ TPU Config","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n    \nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:28.576275Z","iopub.execute_input":"2022-07-20T22:38:28.577069Z","iopub.status.idle":"2022-07-20T22:38:34.500027Z","shell.execute_reply.started":"2022-07-20T22:38:28.577026Z","shell.execute_reply":"2022-07-20T22:38:34.499232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📊 Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_train_essay(essay_id):\n    parent_path = config.base_dir + 'train'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n\ndef get_test_essay(essay_id):\n    parent_path = config.base_dir + 'test'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:34.501528Z","iopub.execute_input":"2022-07-20T22:38:34.501768Z","iopub.status.idle":"2022-07-20T22:38:34.508602Z","shell.execute_reply.started":"2022-07-20T22:38:34.501730Z","shell.execute_reply":"2022-07-20T22:38:34.507556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:34.510322Z","iopub.execute_input":"2022-07-20T22:38:34.510614Z","iopub.status.idle":"2022-07-20T22:38:34.522033Z","shell.execute_reply.started":"2022-07-20T22:38:34.510578Z","shell.execute_reply":"2022-07-20T22:38:34.521307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(config.train_dataset_path)\ndf_test = pd.read_csv(config.test_dataset_path)\ndf_ss = pd.read_csv(config.sample_submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:34.524434Z","iopub.execute_input":"2022-07-20T22:38:34.524933Z","iopub.status.idle":"2022-07-20T22:38:34.981540Z","shell.execute_reply.started":"2022-07-20T22:38:34.524893Z","shell.execute_reply":"2022-07-20T22:38:34.980884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['essay_text'] = df_train['essay_id'].apply(get_train_essay)\ndf_test['essay_text'] = df_test['essay_id'].apply(get_test_essay)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:41:47.786498Z","iopub.execute_input":"2022-07-20T22:41:47.786839Z","iopub.status.idle":"2022-07-20T22:42:36.706932Z","shell.execute_reply.started":"2022-07-20T22:41:47.786794Z","shell.execute_reply":"2022-07-20T22:42:36.706051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_mapping = {\n    'Adequate': 0,\n    'Effective': 1,\n    'Ineffective':2\n}\ndf_train['discourse_effectiveness'] = df_train['discourse_effectiveness'].map(target_mapping) ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:42:46.226671Z","iopub.execute_input":"2022-07-20T22:42:46.227162Z","iopub.status.idle":"2022-07-20T22:42:46.242746Z","shell.execute_reply.started":"2022-07-20T22:42:46.227111Z","shell.execute_reply":"2022-07-20T22:42:46.241931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['discourse_text'] = df_train['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_train['essay_text'] = df_train['essay_text'].apply(resolve_encodings_and_normalize)\n\ndf_test['discourse_text'] = df_test['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_test['essay_text'] = df_test['essay_text'].apply(resolve_encodings_and_normalize)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:42:48.715408Z","iopub.execute_input":"2022-07-20T22:42:48.715699Z","iopub.status.idle":"2022-07-20T22:43:17.279886Z","shell.execute_reply.started":"2022-07-20T22:42:48.715671Z","shell.execute_reply":"2022-07-20T22:43:17.278891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['text'] = df_train['discourse_type'] + \" [SEP] \" + df_train['discourse_text'] + \" [SEP] \" + df_train['essay_text']\ndf_test['text'] = df_test['discourse_type'] + \" [SEP] \" + df_test['discourse_text'] + \" [SEP] \" + df_test['essay_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:43:17.281780Z","iopub.execute_input":"2022-07-20T22:43:17.282038Z","iopub.status.idle":"2022-07-20T22:43:17.372985Z","shell.execute_reply.started":"2022-07-20T22:43:17.282010Z","shell.execute_reply":"2022-07-20T22:43:17.372026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🎟 Tokenizer","metadata":{}},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.model_name)\ntokenizer.save_pretrained(f'{config.save_dir}/tokenizer/')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:43:21.885585Z","iopub.execute_input":"2022-07-20T22:43:21.885899Z","iopub.status.idle":"2022-07-20T22:43:24.615183Z","shell.execute_reply.started":"2022-07-20T22:43:21.885869Z","shell.execute_reply":"2022-07-20T22:43:24.614274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧠 Model","metadata":{}},{"cell_type":"code","source":"\"\"\"class TransformerBlock(Layer):\n    def __init__(self):\n        super(TransformerBlock , self).__init__()\n        self.transformer_model = TFAutoModel.from_pretrained(config.model_name)\n        \n    def call(self,input_tensors):\n        input_id = input_tensors[0]\n        attention_mask = input_tensors[1]\n        transformer_output = self.transformer_model(input_ids = input_id , attention_mask = attention_mask)\n        transformer_output = transformer_output.last_hidden_state\n        return transformer_output\n\nclass ClassificationHead(Layer):\n    def __init__(self):\n        super(ClassificationHead , self).__init__()\n        self.dense = Dense(3, activation='softmax')\n    \n    def call(self , input_tensors):\n        x = self.dense(input_tensors)\n        return x\n\nclass AttentionHead(Layer):\n    def __init__(self):\n        super(AttentionHead , self).__init__()\n        self.dense1 = Dense(512)\n        self.tanh =  tanh\n        self.softmax = softmax\n        self.dense2 = Dense(1,activation=\"softmax\")\n    \n    def call(self , input_tensors):\n        x = self.dense1(input_tensors)\n        x = self.tanh(x)\n        x = self.dense2(x)\n        x = self.softmax(x , axis = 1)\n        return x  \n    \nclass FeedbackPrizeModel(Model):\n    def __init__(self):\n        super(FeedbackPrizeModel, self).__init__()\n        self.transformer_model = TransformerBlock()\n        self.attentionhead = AttentionHead()\n        self.classificationhead = ClassificationHead()\n    \n    def call(self,input_tensors):\n        transformer_output = self.transformer_model(input_tensors)\n        weights = self.attentionhead(transformer_output)\n        context_vector = tf.reduce_sum(weights * transformer_output, axis=1)\n        x = self.classificationhead(context_vector)\n        return x\n    \n    def model(self):\n        input_id = Input(shape = (config.max_length) , dtype = tf.int32, name = 'input_ids')\n        attention_mask = Input(shape = (config.max_length), dtype = tf.int32, name = 'attention_mask')\n        \n        return Model(inputs = [input_id , attention_mask] , outputs = self.call([input_id , attention_mask]))\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:37:51.582660Z","iopub.status.idle":"2022-07-20T22:37:51.583139Z","shell.execute_reply.started":"2022-07-20T22:37:51.582891Z","shell.execute_reply":"2022-07-20T22:37:51.582915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    input_id = Input(shape = (config.max_length) , dtype = tf.int32, name = 'input_ids')\n    attention_mask = Input(shape = (config.max_length), dtype = tf.int32, name = 'attention_mask')\n    \n    transformer_model = TFAutoModel.from_pretrained(config.model_name)\n    cls_token = transformer_model(input_ids = input_id , attention_mask = attention_mask)[0][:,0,:]\n    \n    prediction = Dense(3 , activation = \"softmax\")(cls_token)\n\n    return Model(inputs = [input_id, attention_mask] , outputs = prediction)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:37.766300Z","iopub.execute_input":"2022-07-20T22:38:37.766842Z","iopub.status.idle":"2022-07-20T22:38:37.774947Z","shell.execute_reply.started":"2022-07-20T22:38:37.766794Z","shell.execute_reply":"2022-07-20T22:38:37.773889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving HF Model for inference purpose\ntransformer_model = TFAutoModel.from_pretrained(config.model_name)\ntransformer_model.save_pretrained(f'{config.save_dir}/{config.model_name}')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:38:40.596273Z","iopub.execute_input":"2022-07-20T22:38:40.597056Z","iopub.status.idle":"2022-07-20T22:40:30.218160Z","shell.execute_reply.started":"2022-07-20T22:38:40.596999Z","shell.execute_reply":"2022-07-20T22:40:30.217054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧰 Dataset Prep Function","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef map_function(encodings , target):\n    input_ids = encodings['input_ids']\n    attention_mask = encodings['attention_mask']\n    \n    target = tf.cast(target, tf.int8)\n    \n    return {'input_ids': input_ids , 'attention_mask': attention_mask}, target","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:40:39.530692Z","iopub.execute_input":"2022-07-20T22:40:39.531043Z","iopub.status.idle":"2022-07-20T22:40:39.538005Z","shell.execute_reply.started":"2022-07-20T22:40:39.531013Z","shell.execute_reply":"2022-07-20T22:40:39.536990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔝 Competition Metrics","metadata":{}},{"cell_type":"code","source":"def competition_metrics(y_true, y_preds):\n    return log_loss(y_true, y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:40:41.886104Z","iopub.execute_input":"2022-07-20T22:40:41.886700Z","iopub.status.idle":"2022-07-20T22:40:41.892035Z","shell.execute_reply.started":"2022-07-20T22:40:41.886655Z","shell.execute_reply":"2022-07-20T22:40:41.890901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔄 KFold Training","metadata":{}},{"cell_type":"code","source":"histories = []\nscores = []\nfor fold in range(4,5):\n    print(f\"====== FOLD RUNNING {fold}======\")\n    \n    X_train = df_train.loc[df_train['kfold'] != fold]['text']\n    y_train = df_train.loc[df_train['kfold'] != fold]['discourse_effectiveness']\n    \n    X_test = df_train.loc[df_train['kfold'] == fold]['text']\n    y_test = df_train.loc[df_train['kfold'] == fold]['discourse_effectiveness']\n    \n    print(\" Train Generating Tokens\")\n    train_embeddings = tokenizer(\n        X_train.tolist(),\n        truncation = config.truncation, \n        padding = config.padding,\n        max_length =config.max_length   \n    )\n    \n    print(\" Validation Generating Tokens\")\n    validation_embeddings = tokenizer(\n        X_test.tolist(),\n        truncation = config.truncation, \n        padding = config.padding,\n        max_length =config.max_length   \n    )\n    \n    print(\"Train Generating Dataset\")\n    train = tf.data.Dataset.from_tensor_slices((train_embeddings , y_train))\n    train = (\n                train\n                .map(map_function, num_parallel_calls= config.AUTOTUNE)\n                .batch(config.batch_size)\n                .prefetch(config.AUTOTUNE)\n            )\n        \n    print(\"Validation Generating Dataset\")\n    val = tf.data.Dataset.from_tensor_slices((validation_embeddings , y_test))\n    val = (\n                val\n                .map(map_function, num_parallel_calls= config.AUTOTUNE)\n                .batch(config.batch_size)\n                .prefetch(config.AUTOTUNE)\n            )\n    \n    #Clearing backend session\n    K.clear_session()\n    print(\"Backend Cleared\")\n\n    print(\"Model Creation\")\n    with strategy.scope():\n        model = create_model()\n        model.compile(\n          optimizer = Adam(learning_rate = config.learning_rate), \n          metrics = ['accuracy'],\n          loss = SparseCategoricalCrossentropy()\n      )    \n    early_stopping=EarlyStopping(monitor=\"val_loss\",min_delta=0,patience=4,verbose=1,mode=\"min\",restore_best_weights=True)\n    \n    hist = model.fit(train , validation_data = val , epochs = config.epochs, callbacks = [early_stopping])\n    \n    # prediction on val\n    print(\"prediction on validation data\")\n    preds = model.predict(val , verbose = 1)\n    score = competition_metrics(y_test.values,preds)\n    scores.append(score)\n    print(f\"Log Loss for Fold {fold} is {score}\")\n\n    #saving model\n    print(\"saving model\")\n    \n    localhost_save_option = tf.saved_model.SaveOptions(experimental_io_device=\"/job:localhost\")\n    model.save(f'{config.save_dir}/{config.model_name}_{fold}', options=localhost_save_option)\n    \n    del model, X_train , y_train, X_test, y_test, val , train , train_embeddings , validation_embeddings\n\n    histories.append(hist)\n\nprint(\"the final average Log Loss is \", np.mean(scores))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T22:43:39.048063Z","iopub.execute_input":"2022-07-20T22:43:39.048638Z","iopub.status.idle":"2022-07-20T23:28:06.644862Z","shell.execute_reply.started":"2022-07-20T22:43:39.048596Z","shell.execute_reply":"2022-07-20T23:28:06.643946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}