{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import log_loss\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras.models import Model\n\nfrom transformers import AutoTokenizer, TFAutoModel\n\ntqdm.pandas()\nnp.random.seed(2022)\ntf.random.set_seed(2022)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:01.463515Z","iopub.execute_input":"2022-08-07T18:05:01.464008Z","iopub.status.idle":"2022-08-07T18:05:10.441523Z","shell.execute_reply.started":"2022-08-07T18:05:01.463894Z","shell.execute_reply":"2022-08-07T18:05:10.440863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hardware Config","metadata":{}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    BATCH_SIZE = strategy.num_replicas_in_sync * 8\n    print(\"Running on TPU:\", tpu.master())\n    print(f\"Batch Size: {BATCH_SIZE}\")\n    \nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\n    BATCH_SIZE = 64\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    print(f\"Batch Size: {BATCH_SIZE}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:10.442809Z","iopub.execute_input":"2022-08-07T18:05:10.443200Z","iopub.status.idle":"2022-08-07T18:05:16.550068Z","shell.execute_reply.started":"2022-08-07T18:05:10.443167Z","shell.execute_reply":"2022-08-07T18:05:16.549159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hyperparameters","metadata":{}},{"cell_type":"code","source":"class Config:\n    \n    FOLDS = 5\n    VERBOSE = 0\n    NUM_EPOCH = 7\n    MAX_LEN = 512\n    LR_START = 8e-6\n    LR_END = 1e-7\n    BATCH_SIZE = BATCH_SIZE\n    MODEL_NAME = 'roberta-large'\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:16.551277Z","iopub.execute_input":"2022-08-07T18:05:16.551536Z","iopub.status.idle":"2022-08-07T18:05:16.557042Z","shell.execute_reply.started":"2022-08-07T18:05:16.551506Z","shell.execute_reply":"2022-08-07T18:05:16.555848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load training dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/feedback-train-dataset-with-folds/train.csv')\nprint(f\"train: {train.shape}\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:16.558949Z","iopub.execute_input":"2022-08-07T18:05:16.559236Z","iopub.status.idle":"2022-08-07T18:05:18.800417Z","shell.execute_reply.started":"2022-08-07T18:05:16.559205Z","shell.execute_reply":"2022-08-07T18:05:18.799283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['final_text'] = train.progress_apply(lambda x: f\"{x['discourse_type']} {x['discourse_text']} </s> {x['essay_text']}\", axis=1)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:18.801769Z","iopub.execute_input":"2022-08-07T18:05:18.802199Z","iopub.status.idle":"2022-08-07T18:05:19.867323Z","shell.execute_reply.started":"2022-08-07T18:05:18.802166Z","shell.execute_reply":"2022-08-07T18:05:19.866671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build the model","metadata":{}},{"cell_type":"code","source":"def encode_text(text, tokenizer):\n    \n    encoded = tokenizer.batch_encode_plus(\n        text,\n        add_special_tokens=True,\n        max_length=config.MAX_LEN,\n        padding='max_length',\n        truncation=True,\n        return_attention_mask=True,\n        return_tensors=\"tf\",\n    )\n\n    input_ids = np.array(encoded[\"input_ids\"], dtype=\"int32\")\n    attention_masks = np.array(encoded[\"attention_mask\"], dtype=\"int32\")\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_masks\": attention_masks\n    }","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:19.868729Z","iopub.execute_input":"2022-08-07T18:05:19.868970Z","iopub.status.idle":"2022-08-07T18:05:19.874860Z","shell.execute_reply.started":"2022-08-07T18:05:19.868941Z","shell.execute_reply":"2022-08-07T18:05:19.873925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feedback_model(transformer_model):\n    \n    input_ids = layers.Input(shape=(config.MAX_LEN,), dtype=tf.int32, name=\"input_ids\")\n    attention_mask = layers.Input(shape=(config.MAX_LEN,), dtype=tf.int32, name=\"attention_mask\")\n\n    bert_model = transformer_model(input_ids, attention_mask=attention_mask)\n    \n    last_hidden_state, pooler_output = bert_model[0], bert_model[1]\n    \n    x = layers.Concatenate()([\n        pooler_output,\n        layers.GlobalAveragePooling1D()(last_hidden_state),\n        layers.GlobalMaxPooling1D()(last_hidden_state)\n    ])\n    x = layers.Dropout(rate=0.35)(x)\n    \n    x = layers.Dense(units=1024, activation='gelu')(x)\n    x = layers.Dropout(rate=0.25)(x)\n    \n    x_output = layers.Dense(units=7, activation='softmax')(x)\n\n    model = Model(inputs=[input_ids, attention_mask], \n                  outputs=x_output, \n                  name='Feedback_TFRoberta_Large_Model')\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:19.876257Z","iopub.execute_input":"2022-08-07T18:05:19.876541Z","iopub.status.idle":"2022-08-07T18:05:19.887118Z","shell.execute_reply.started":"2022-08-07T18:05:19.876511Z","shell.execute_reply":"2022-08-07T18:05:19.886170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_head(transformer_output):\n    \n    re = layers.Reshape((8, 8, 16))(transformer_output)\n    \n    x1 = layers.Conv2D(filters=32, kernel_size=3, strides=1,  \n                       padding='same', activation='gelu')(re)\n    x2 = layers.SpatialDropout2D(rate=0.2)(x1)\n    \n    x2 = layers.Conv2D(filters=48, kernel_size=3, strides=1, \n                       padding='same', activation='gelu')(x2)\n    x3 = layers.SpatialDropout2D(rate=0.2)(x2)\n    \n    x3 = layers.Conv2D(filters=64, kernel_size=3, strides=1, \n                       padding='same', activation='gelu')(x3)\n    \n    x4 = layers.MaxPool2D()(x3)\n    x4 = layers.Flatten()(x4)\n    \n    add0 = layers.Add()([transformer_output, x4])\n    \n    x_output = layers.Average()([\n        layers.Dense(units=3, activation='softmax')(layers.Dropout(rate=0.35)(add0)),\n        layers.Dense(units=3, activation='softmax')(layers.Dropout(rate=0.25)(add0)),\n        layers.Dense(units=3, activation='softmax')(layers.Dropout(rate=0.15)(add0))\n    ])\n    return x_output","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:08:50.308760Z","iopub.execute_input":"2022-08-07T18:08:50.310751Z","iopub.status.idle":"2022-08-07T18:08:50.322895Z","shell.execute_reply.started":"2022-08-07T18:08:50.310677Z","shell.execute_reply":"2022-08-07T18:08:50.321942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.MODEL_NAME)\ntokenizer.save_pretrained(f'./{config.MODEL_NAME}-tokenizer')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:19.906638Z","iopub.execute_input":"2022-08-07T18:05:19.907184Z","iopub.status.idle":"2022-08-07T18:05:22.760130Z","shell.execute_reply.started":"2022-08-07T18:05:19.907138Z","shell.execute_reply":"2022-08-07T18:05:22.759239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer_model = TFAutoModel.from_pretrained(config.MODEL_NAME)\ntransformer_model.save_pretrained(f'./{config.MODEL_NAME}-model')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:05:22.762629Z","iopub.execute_input":"2022-08-07T18:05:22.762881Z","iopub.status.idle":"2022-08-07T18:06:08.539584Z","shell.execute_reply.started":"2022-08-07T18:05:22.762848Z","shell.execute_reply":"2022-08-07T18:06:08.538369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pretrained_model = feedback_model(transformer_model)\nmodel = Model(inputs=pretrained_model.input, \n              outputs=model_head(pretrained_model.get_layer(\n                  pretrained_model.layers[-3].name\n              ).output),\n              name='Feedback_TFRoberta_Large_Model')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:08:53.838236Z","iopub.execute_input":"2022-08-07T18:08:53.838556Z","iopub.status.idle":"2022-08-07T18:08:57.607212Z","shell.execute_reply.started":"2022-08-07T18:08:53.838521Z","shell.execute_reply":"2022-08-07T18:08:57.606236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model, to_file='./Feedback_TFRoberta_Large_Model.png', \n    show_shapes=True, show_layer_names=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:09:00.088651Z","iopub.execute_input":"2022-08-07T18:09:00.089268Z","iopub.status.idle":"2022-08-07T18:09:00.699005Z","shell.execute_reply.started":"2022-08-07T18:09:00.089229Z","shell.execute_reply":"2022-08-07T18:09:00.697903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    \n    tokenizer = AutoTokenizer.from_pretrained(config.MODEL_NAME)\n    transformer_model = TFAutoModel.from_pretrained(config.MODEL_NAME)\n    pretrained_model = feedback_model(transformer_model)\n    \n    counter = 0\n    oof_score = 0\n    loss_values = {}\n    \n\n    for fold in range(config.FOLDS):\n        counter += 1\n\n        train_data = encode_text(train[train['kfold']!=fold]['final_text'].tolist(), tokenizer)\n        val_data = encode_text(train[train['kfold']==fold]['final_text'].tolist(), tokenizer)\n        \n        train_labels, val_labels = pd.get_dummies(train[train['kfold']!=fold]['discourse_effectiveness']), \\\n                                   pd.get_dummies(train[train['kfold']==fold]['discourse_effectiveness'])\n\n        pretrained_model.load_weights(f'../input/feedback-roberta-large-pretraining-v2-2/Feedback_TFRoberta_Large_Model_{counter}C.h5')\n        \n        model = Model(inputs=pretrained_model.input, \n                      outputs=model_head(pretrained_model.get_layer(\n                          pretrained_model.layers[-3].name\n                      ).output),\n                      name='Feedback_TFRoberta_Large_Model')\n        \n        model.compile(loss='categorical_crossentropy', \n                      optimizer=optimizers.Adam(learning_rate=config.LR_START))\n\n        early = callbacks.EarlyStopping(monitor=\"val_loss\", mode=\"min\", \n                                        restore_best_weights=True, \n                                        patience=4, verbose=config.VERBOSE)\n\n        reduce_lr = callbacks.ReduceLROnPlateau(monitor=\"val_loss\", factor=0.5, \n                                                min_lr=config.LR_END, patience=1, \n                                                verbose=config.VERBOSE, mode='min')\n\n        chk_point = callbacks.ModelCheckpoint(f'./Feedback_TFRoberta_Large_Model_{counter}C.h5', \n                                              monitor='val_loss', verbose=config.VERBOSE, \n                                              save_best_only=True, mode='min',\n                                              save_weights_only=True)\n\n        history = model.fit(\n            (np.asarray(train_data['input_ids']),\n             np.asarray(train_data['attention_masks'])), \n            train_labels, \n            batch_size=config.BATCH_SIZE,\n            epochs=config.NUM_EPOCH, \n            verbose=config.VERBOSE, \n            callbacks=[reduce_lr, early, chk_point], \n            validation_data=(\n                (np.asarray(val_data['input_ids']),\n                 np.asarray(val_data['attention_masks'])), \n                val_labels\n            )\n        )\n\n        loss_values[\"train_loss_\"+str(counter)] = history.history['loss']\n        loss_values[\"valid_loss_\"+str(counter)] = history.history['val_loss']\n        \n        model.load_weights(f'./Feedback_TFRoberta_Large_Model_{counter}C.h5')\n\n        y_pred = model.predict(\n            (np.asarray(val_data['input_ids']),\n             np.asarray(val_data['attention_masks'])), \n            batch_size=config.BATCH_SIZE, \n            verbose=config.VERBOSE\n        )\n        \n        score = log_loss(val_labels, y_pred)\n        oof_score += score\n        print(f\"Fold-{counter} | OOF Score: {score}\")\n        \n        del history\n        del model, y_pred  \n        del val_data, val_labels\n        del train_data, train_labels\n        gc.collect()\n\n\noof_score /= float(counter)\nprint(f\"Aggregate OOF Score: {oof_score}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T08:38:02.853207Z","iopub.execute_input":"2022-08-07T08:38:02.855761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 7))\nplt.title(\"Model Loss Curve\", fontweight='bold', pad=15)\n\nfor i in range(config.FOLDS):\n    plt.plot(loss_values[\"train_loss_\"+str(i+1)], label='train_loss_'+str(i+1))\n    plt.plot(loss_values[\"valid_loss_\"+str(i+1)], label='valid_loss_'+str(i+1))\n\nplt.ylabel('Model Loss')\nplt.xlabel('Epochs')\nplt.legend()\nplt.grid();","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Good Day!!","metadata":{},"execution_count":null,"outputs":[]}]}