{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n<h1 style = \"font-size:60px; font-weight : normal; background-color: #f6f5f5 ; color : #123456; text-align: center; border-radius: 100px 100px;\">Feedback Prize - Predicting Effective Arguments</h1>\n<br>\n<h2 style = \"background-color: #f6f5f5 ; color : #123456; text-align: center;\">DeBERTa-v3 Type - Text - Essay</h2>\n<br>\n<img src=\"https://storage.googleapis.com/kaggle-media/competitions/The%20Learning%20Agency/Kaggle%20Description%20Image.png\" width=\"500\" height=\"600\">","metadata":{}},{"cell_type":"markdown","source":"<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Simple Approach:<br></span>\n* Concatenate <code>Discourse Type</code> <code>[SEP]</code> <code>Discourse Text</code> <code>[SEP]</code> <code>Essay</code>\n* Pass through Deberta-base-v3 model\n* Apply Weighted Pooling for last 6 layer outputs of DeBERTa and Mean Pooling on the combination\n* Using Multi Sample Dropout with 8 samples and start_prob=0.2 with increment=0.01\n* Classify using <code>CrossEntropyLoss</code> into 3 classes\n\nV9. Change Weighted Pooling to Context Pooling\n\nApply Stratified Group Kfold with k-3 to save time and resources <br>\n\nTraining using many optimization approaches like: Freeze, Gradient Accumulation, Gradient Checkpoint, Adam8bit","metadata":{}},{"cell_type":"markdown","source":"# Install Required Libraries","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade wandb\n!pip install --upgrade transformers\n!pip install -q bitsandbytes-cuda110\n!pip install sentencepiece\n# !pip install codecarbon\n# !pip install torchsummary ","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T18:44:42.214192Z","iopub.execute_input":"2022-08-02T18:44:42.214965Z","iopub.status.idle":"2022-08-02T18:45:39.885530Z","shell.execute_reply.started":"2022-08-02T18:44:42.214865Z","shell.execute_reply":"2022-08-02T18:45:39.884198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Required Libraries","metadata":{}},{"cell_type":"code","source":"# import manipulation\nimport numpy as np\nimport pandas as pd\n\n# import Pytorch\nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\n\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.utils.checkpoint import checkpoint\nfrom torch.autograd import Variable\n\n# import wandb\nimport wandb\n\nimport tokenizers\n\n# import Transformer model\nimport transformers\nfrom transformers import AutoTokenizer, AutoModel, AutoConfig, AdamW\nfrom transformers import DataCollatorWithPadding\nfrom transformers.models.deberta_v2.modeling_deberta_v2 import StableDropout, ContextPooler\n\n# import SKLearn\nfrom sklearn.model_selection import  KFold, GroupKFold, StratifiedKFold, StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import log_loss\n\n\n# import ...\nimport string\nimport random\nimport os\nimport joblib\nimport gc\nimport copy\nimport time\n\n\n# other\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n#8-bits optimizer\nimport bitsandbytes as bnb\n\n# from codecarbon import track_emissions\n\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:39.889154Z","iopub.execute_input":"2022-08-02T18:45:39.889891Z","iopub.status.idle":"2022-08-02T18:45:47.758216Z","shell.execute_reply.started":"2022-08-02T18:45:39.889859Z","shell.execute_reply":"2022-08-02T18:45:47.757184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"torch.__version__: {torch.__version__}\")\nprint(f\"tokenizers.__version__: {tokenizers.__version__}\")\nprint(f\"transformers.__version__: {transformers.__version__}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:47.760191Z","iopub.execute_input":"2022-08-02T18:45:47.760875Z","iopub.status.idle":"2022-08-02T18:45:47.772714Z","shell.execute_reply.started":"2022-08-02T18:45:47.760836Z","shell.execute_reply":"2022-08-02T18:45:47.771691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://i.imgur.com/gb6B4ig.png\" width=\"400\" alt=\"Weights & Biases\" />\n\n<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\"> Weights & Biases (W&B) is a set of machine learning tools that helps you build better models faster. <strong>Kaggle competitions require fast-paced model development and evaluation</strong>. There are a lot of components: exploring the training data, training different models, combining trained models in different combinations (ensembling), and so on.</span>\n\n> <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">⏳ Lots of components = Lots of places to go wrong = Lots of time spent debugging</span>\n\n<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">W&B can be useful for Kaggle competition with it's lightweight and interoperable tools:</span>\n\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Quickly track experiments,<br></span>\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Version and iterate on datasets, <br></span>\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Evaluate model performance,<br></span>\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Reproduce models,<br></span>\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Visualize results and spot regressions,<br></span>\n* <span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Share findings with colleagues.</span>\n\n<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">To learn more about Weights and Biases check out this <strong><a href=\"https://www.kaggle.com/ayuraj/experiment-tracking-with-weights-and-biases\">kernel</a></strong>.</span>\n\n<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.2em; font-weight: 300;\">Go to Add-ons -> Secrets and provide your Wandb access token with Label name as wandb_api and value from <strong><a href=\"https://wandb.ai/authorize\">https://wandb.ai/authorize</a></strong>.</span>","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\n\n# Get secret key from kaggle\n# Go to Add-ons -> Secrets and provide your Wandb access token with Label name as wandb_api and value from https://wandb.ai/authorize\nuser_client = UserSecretsClient()\napi_key = user_client.get_secret(\"wandb_api\")\n\n# Connect to wandb\nwandb.login(key=api_key)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:47.776410Z","iopub.execute_input":"2022-08-02T18:45:47.776741Z","iopub.status.idle":"2022-08-02T18:45:48.686729Z","shell.execute_reply.started":"2022-08-02T18:45:47.776714Z","shell.execute_reply":"2022-08-02T18:45:48.685460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 2022\n    max_length = 512\n    epoch = 4\n    train_batch_size = 16\n    valid_batch_size = 32\n\n    model_name = \"microsoft/deberta-v3-base\"\n    token_name = \"microsoft/deberta-v3-base\"\n\n    scheduler = \"CosineAnnealingLR\"\n    learning_rate = 1e-5\n    min_lr = 1e-6\n    T_max = 500\n    weight_decay = 0.005\n    dropout = 0.1\n    \n    num_classes = 3\n    n_fold = 3\n    n_accumulate = 2\n    freezing = True\n    gradient_checkpoint = True\n    device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n    \n    wandb_id = f\"PL{round(time.time())}\" # ID on WandB\n    group = f'{wandb_id}-Baseline'\n    competition = \"FeedBack\"\n    _wandb_kernel = \"deb\"\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.token_name, use_fast=False)\nCFG.tokenizer.model_max_length = CFG.max_length\nCFG.tokenizer.is_fast","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:48.688356Z","iopub.execute_input":"2022-08-02T18:45:48.689928Z","iopub.status.idle":"2022-08-02T18:45:51.185583Z","shell.execute_reply.started":"2022-08-02T18:45:48.689888Z","shell.execute_reply":"2022-08-02T18:45:51.184420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AutoConfig.from_pretrained(CFG.model_name)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T18:45:51.187110Z","iopub.execute_input":"2022-08-02T18:45:51.188132Z","iopub.status.idle":"2022-08-02T18:45:51.299728Z","shell.execute_reply.started":"2022-08-02T18:45:51.188093Z","shell.execute_reply":"2022-08-02T18:45:51.298777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils\n\nSet Seed for Reproducibility","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    \nseed_everything(seed=CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:51.301083Z","iopub.execute_input":"2022-08-02T18:45:51.301514Z","iopub.status.idle":"2022-08-02T18:45:51.311895Z","shell.execute_reply.started":"2022-08-02T18:45:51.301473Z","shell.execute_reply":"2022-08-02T18:45:51.310694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def criterion(outputs, labels):\n    \"\"\"\n    Calculate Cross Entropy Loss\n    \"\"\"\n    return nn.CrossEntropyLoss()(outputs, labels)\n\ndef get_score(outputs, labels):\n    \"\"\"\n    Calculate Log Loss from softmax output\n    \"\"\"\n    outputs = F.softmax(torch.tensor(outputs)).numpy()\n    return log_loss(labels, outputs)\n\ndef freeze(module):\n    \"\"\"\n    Freezes module's parameters.\n    \"\"\"\n    for parameter in module.parameters():\n        parameter.requires_grad = False\n\ndef get_freezed_parameters(module):\n    \"\"\"\n    Returns names of freezed parameters of the given module.\n    \"\"\"\n    freezed_parameters = []\n    for name, parameter in module.named_parameters():\n        if not parameter.requires_grad:\n            freezed_parameters.append(name)\n            \n    return freezed_parameters\n\n# 8-bits optimizer\ndef set_embedding_parameters_bits(embeddings_path, optim_bits=32):\n    \"\"\"\n    https://github.com/huggingface/transformers/issues/14819#issuecomment-1003427930\n    \"\"\"\n    embedding_types = (\"word\", \"position\", \"token_type\")\n    for embedding_type in embedding_types:\n        attr_name = f\"{embedding_type}_embeddings\"\n        \n        if hasattr(embeddings_path, attr_name): \n            bnb.optim.GlobalOptimManager.get_instance().register_module_override(\n                getattr(embeddings_path, attr_name), 'weight', {'optim_bits': optim_bits}\n            )","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:51.315532Z","iopub.execute_input":"2022-08-02T18:45:51.316134Z","iopub.status.idle":"2022-08-02T18:45:51.329693Z","shell.execute_reply.started":"2022-08-02T18:45:51.316103Z","shell.execute_reply":"2022-08-02T18:45:51.328641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-07T03:34:14.301283Z","iopub.execute_input":"2022-07-07T03:34:14.301749Z","iopub.status.idle":"2022-07-07T03:34:14.306954Z","shell.execute_reply.started":"2022-07-07T03:34:14.301711Z","shell.execute_reply":"2022-07-07T03:34:14.306164Z"}}},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/feedback-prize-effectiveness\"\nTRAIN_DIR = os.path.join(INPUT_DIR, \"train\")\nTRAIN_CSV = os.path.join(INPUT_DIR, \"train.csv\")\n\nTEST_DIR = os.path.join(INPUT_DIR, \"test\")\nTEST_CSV = os.path.join(INPUT_DIR, \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:51.333403Z","iopub.execute_input":"2022-08-02T18:45:51.334252Z","iopub.status.idle":"2022-08-02T18:45:51.344749Z","shell.execute_reply.started":"2022-08-02T18:45:51.334214Z","shell.execute_reply":"2022-08-02T18:45:51.343491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read the Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:51.350130Z","iopub.execute_input":"2022-08-02T18:45:51.351240Z","iopub.status.idle":"2022-08-02T18:45:51.643521Z","shell.execute_reply.started":"2022-08-02T18:45:51.351202Z","shell.execute_reply":"2022-08-02T18:45:51.642370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get Essay text for each rows\ndef get_essay(essay_id):\n    path = os.path.join(TRAIN_DIR, f\"{essay_id}.txt\")\n    essay_text = open(path, 'r').read()\n    return essay_text\n    \ndf['essay_text'] = df['essay_id'].apply(get_essay)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:45:51.645572Z","iopub.execute_input":"2022-08-02T18:45:51.645993Z","iopub.status.idle":"2022-08-02T18:46:21.522165Z","shell.execute_reply.started":"2022-08-02T18:45:51.645952Z","shell.execute_reply":"2022-08-02T18:46:21.521183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:21.523637Z","iopub.execute_input":"2022-08-02T18:46:21.524754Z","iopub.status.idle":"2022-08-02T18:46:21.682412Z","shell.execute_reply.started":"2022-08-02T18:46:21.524711Z","shell.execute_reply":"2022-08-02T18:46:21.681341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:21.683904Z","iopub.execute_input":"2022-08-02T18:46:21.685025Z","iopub.status.idle":"2022-08-02T18:46:21.693608Z","shell.execute_reply.started":"2022-08-02T18:46:21.684981Z","shell.execute_reply":"2022-08-02T18:46:21.692507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_text'].str.split(\" \").apply(len).describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:21.694906Z","iopub.execute_input":"2022-08-02T18:46:21.695992Z","iopub.status.idle":"2022-08-02T18:46:22.187666Z","shell.execute_reply.started":"2022-08-02T18:46:21.695948Z","shell.execute_reply":"2022-08-02T18:46:22.186463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:22.189719Z","iopub.execute_input":"2022-08-02T18:46:22.190124Z","iopub.status.idle":"2022-08-02T18:46:22.199915Z","shell.execute_reply.started":"2022-08-02T18:46:22.190085Z","shell.execute_reply":"2022-08-02T18:46:22.198528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lowercasing","metadata":{}},{"cell_type":"code","source":"df['essay_text'] = df['essay_text'].str.lower()\ndf['discourse_text'] = df['discourse_text'].str.lower()\ndf['discourse_type'] = df['discourse_type'].str.lower()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:22.202347Z","iopub.execute_input":"2022-08-02T18:46:22.202831Z","iopub.status.idle":"2022-08-02T18:46:22.512995Z","shell.execute_reply.started":"2022-08-02T18:46:22.202793Z","shell.execute_reply":"2022-08-02T18:46:22.511949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Normalize unicode","metadata":{}},{"cell_type":"code","source":"from text_unidecode import unidecode\nfrom typing import Dict, List, Tuple\nimport codecs\n\ndef replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:22.514587Z","iopub.execute_input":"2022-08-02T18:46:22.514967Z","iopub.status.idle":"2022-08-02T18:46:22.534977Z","shell.execute_reply.started":"2022-08-02T18:46:22.514931Z","shell.execute_reply":"2022-08-02T18:46:22.533934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['discourse_text'] = df['discourse_text'].apply(lambda x : resolve_encodings_and_normalize(x))\ndf['essay_text'] = df['essay_text'].apply(lambda x : resolve_encodings_and_normalize(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:22.536427Z","iopub.execute_input":"2022-08-02T18:46:22.537031Z","iopub.status.idle":"2022-08-02T18:46:49.639858Z","shell.execute_reply.started":"2022-08-02T18:46:22.536989Z","shell.execute_reply":"2022-08-02T18:46:49.638712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Encode target label to numeric","metadata":{}},{"cell_type":"code","source":"# Convert discourse_effectiveness cate value to numeric 0-1-2\nencoder = LabelEncoder()\ndf['discourse_effectiveness'] = encoder.fit_transform(df['discourse_effectiveness'])\ndf['discourse_type'] = encoder.fit_transform(df['discourse_type'])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.641734Z","iopub.execute_input":"2022-08-02T18:46:49.642169Z","iopub.status.idle":"2022-08-02T18:46:49.676305Z","shell.execute_reply.started":"2022-08-02T18:46:49.642110Z","shell.execute_reply":"2022-08-02T18:46:49.675265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.677815Z","iopub.execute_input":"2022-08-02T18:46:49.678963Z","iopub.status.idle":"2022-08-02T18:46:49.695336Z","shell.execute_reply.started":"2022-08-02T18:46:49.678920Z","shell.execute_reply":"2022-08-02T18:46:49.694109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Stratified KFold\n\nSplit Train df by StratifiedKFold with class is discourse_effectiveness.\n\nWith StratifiedGroupKFold, groups is essay_id\n\nIn experiments, StratifiedKFold has better result than StratifiedGroupKFold on CV and LB.\n\nMaybe because of label leakage???\n\n![image.png](attachment:26773cb2-16a4-4276-bda0-d64a705b52ba.png)!","metadata":{},"attachments":{"26773cb2-16a4-4276-bda0-d64a705b52ba.png":{"image/png":"iVBORw0KGgoAAAANSUhEUgAAAlgAAAEsCAIAAACQX1rBAAAgAElEQVR4nO3dd1xTV/8H8HMJZEGMDFkyXQyRKiBVcA9wY90WK7jqQkXEQfuoaLVorVhtFa2PBW2dLW6pigNEcTIUFRGVYRUXCkGBBEh+f9yaXx5AhBgyyOf98o+bc0/u/Z5Lkq/33nPOpSQSCQEAANBWOqoOAAAAQJWQCAEAQKshEQIAgFZDIgQAAK2GRAgAAFoNiRAAALQaEiEAAGg1JEIAANBqSIQAAKDVkAgBAECrIRECAIBWQyIEAACthkQIAABaDYkQAAC0GhKhBouPjx8+fLi5uTmTyTQ2NnZ2dvb399++fbtIJFJJPAkJCRRFBQYGyha+fPly4sSJFhYWDAaDoqiYmBhCCEVRdnZ2itrvh7b25MmT//znP56eniYmJnp6ekZGRp9//vmCBQtu3LihqF03BoqiKIqqVlhZWTlq1CiKotq1a/fkyRNCSExMDPUB9EGuJ3o74eHhdVfLzc2lKKpXr14NbQ6A+tNVdQAgp+XLl69cuZIQ4uLi4u3tzWAwsrKy9u7du2fPnqFDh5qbmxNCKIqytbXNzc1V7K7Dw8NXrFgRHR1dLefVasqUKceOHXN1de3bt6+urm6bNm0UG8yHbNu2LTg4uLy83MjIyNPT08jIqLi4OD09PTIyMjIycu7cuRs3blROJJ+uoqJi3LhxBw8edHBwOHfunKWlpXRV69atu3XrVq2+0g4yQNOARKiRbty4sXLlSiaTeejQoUGDBknLnzx5sn37dhaLpZKoPD09MzMz+Xy+tEQkEsXFxdnZ2aWlpeno/P/lh8zMTD09vcaLZNu2bTNmzODxeNu2bZswYYLsrq9cubJq1aqsrKzG27tiVVRUjB079tChQ46OjufPn6f/iyPVrVu3Bp3/AUBNSIQa6dChQ4SQMWPGyGZBQkjLli0/eo2r8XC5XEdHR9mSZ8+eVVVV2drayqYiQki1aor1+PHjefPmMRiMuLi4mmdLXbp0OX78eEpKSuMFoEAVFRWjR48+cuSIs7PzuXPnzMzMVB0RQBOEe4Qa6eXLl4SQFi1afKgCfeOHEJKXlye9dSS9wWNnZ0dRlEQi+fnnnz/77DMul9uxY0dCiEQi2bt377hx49q1a6evr8/j8Tw9Pbds2SIWi6VbtrOzW7FiBSFk0qRJ0i0nJCSQGvcI7ezsbG1tCSGJiYl0NemdvFrv6mVkZPj7+7ds2ZLFYllaWk6aNKnmRd13794tXrzYxsaGzWY7OjpGRkZKJJJqdTZt2iQUCsePH18zC0q5u7tXO1bh4eH3798fN26cmZmZjo7O4cOH6bVxcXH9+/c3NDRks9kODg5LliwpKiqS3VRgYKD0CMiq1kbZvYwcOdLY2FhfX9/b2zsuLu5DQYpEolGjRh05csTFxeX8+fPyZcGPxl+rV69eTZ8+3dzcnMvldurUadeuXXLsGkBT4IxQI1lZWRFCYmNjw8LCak2Hbdq0CQgI2Llzp76+/qhRo+jCaudhM2bMiI6O7tmzp5OTE92/RigUfvnll4aGhs7Ozm5ubq9evbp8+fLs2bOvXbsmvf42atSoM2fO3Lx509vbW3ovqtr1OmnN3Nzc2NhYMzOzAQMGEEJMTEw+1KLY2Ngvv/xSJBK5u7t7eXk9fPgwJibm2LFjiYmJ7du3p+sIhUIfH5/k5GQTE5OhQ4eWlJQsWbLk4cOH1Tb1999/E0LGjh370cMoKysrq3PnzsbGxr17937z5g195TYiIuKbb77R1dXt2bOniYnJpUuX1q5de+jQoQsXLsiXlh4+fEjfsPTx8Xn69GlSUtKQIUN+++23mndbRSLRyJEjjx8/7urqevbs2ToOXR3ki7+wsNDb2/v+/ftWVlbDhg179uzZpEmTZsyYIUcAAJpBAhrowYMHbDabENKsWbOJEydu37799u3bYrG4WjVCiK2tbc230ydqJiYmt2/fli2vqKiIjY0VCoXSkhcvXnh4eBBCEhMTpYXLly8nhERHR1fb7Pnz5wkhAQEB0pKcnBxCSM+ePesO7NGjR1wul8/ny+5l586dhJDOnTtLS77//ntCiKenZ1FREV2SkpLSrFkz2a2JRCL6VPiff/6p2fBaRUdH09+FoKCgyspKafm1a9d0dHR4PN7Vq1fpkvLy8tGjRxNCRo8eLa0WEBBACDl//nzdbZTuZeLEiRUVFXThsWPHGAyGvr7+06dPZd9ICBk8eDAhpGPHjq9evaojbNmjXU0946e3s3z5cmnJ119/TQjx8/MrLy+nS+Li4nR1dWv9UwI0AUiEmurUqVOyvQcJIaampgsXLnzz5o20Tt2JcN26dfXZUXx8PCEkJCREWqLwRDhv3jxCyLZt26pVGz58OCEkJSWFfmltbU0IuXTpkmydsLAw2a09e/aMPhrSH3FaZmZmwP+SZlM6E7Ro0eLdu3eyb5k4cSIhZOnSpbKFz58/53A4Ojo60kTboERoYGDw+vVr2Wr0mev3338v+0YaRVEZGRmSD5Bm1mr8/PwaFH+1RFhSUsLhcHR1dfPy8mTfOH78eCRCaKpwaVRT+fj4PHr06OjRo/Hx8VevXr19+/aLFy/WrVt36NCh5OTkOm4fSg0bNqzW8vT09NOnT+fl5ZWWlkokkpKSEkJIdna2ghsgg861fn5+1cq7det2+PDh69evu7m55efnP378uGXLll5eXrJ1xo8fHxERIX0pqXHLkPbs2TP6FFNqzZo1sh1c+/Xrx+VyZSskJSURQvz9/WULTU1NfXx8jhw5kpycTJ9dNYiPj4+hoWG1+Pfv33/x4sVqNb29vS9dujRu3LjExERjY+MPbbDm8Ak3N7dPiT81NbWsrMzb29vGxqZanHv37v14CwE0EBKhBmOxWKNHj6Z/zl6+fBkTExMeHv7gwYNvvvlm+/btH317tV86QohIJAoMDKz1945Oh42E7hRT641GQsirV68IIU+fPiW1xVytxNjYmO4HVFhYKHvG3KtXL2mOtLOzy8vLq3s79B7pgZjVyukuMHQ8DVX/rZ04caJPnz6pqam+vr7nzp2jrwDXVMfwCfnir+dxBmhKkAibiBYtWixcuJDD4cyZM+fEiRP1eQt9l1FWZGTk3r17XVxc1q1b5+bmZmhoqKend//+fQcHhw+daSlEVVUVRVH0pbya6M4ydAA1p1ypVqKnp+fk5HT37t3U1NRql47rVvNo1K1mJLJk+9nW7UMHls/nnzp1qmfPnikpKUOGDDl58mS1E9ZP9KH4P3ScAZowJMImhR4gQZ9CyYEenkjnQmnho0ePFBJbHaysrB4+fLhp06YPnfcQQuisVvNMrmbJwIED7969u3///iFDhnxKVJaWljk5OXl5eQ4ODjX3aGFhQb9kMpmEkLdv38rWefz4ca3brBltfn4+ed+6akxMTOLj47t3756UlDRixIijR4/S+1Js/DXfVUecAE0SxhFqpA+dRtBjCaS/qnp6epWVlfXf7Js3bwghdJ8UqQMHDlSrRv8cN2jLdevXrx8hRDp0r1a2trZWVlZPnjy5fPmybPm+ffuq1ZwzZw6Tydy7d2/NG28N0r17d0LI7t27ZQtfvnx5+vRpHR0d6a1KOqPcv39fttrp06dr3ebp06erDeOjL0R7e3vXWt/S0vLs2bMtW7Y8derU+PHjq6qqFB5/Ne7u7mw2++rVq9Vyec3jDNBkIBFqpKVLly5atIjukymVnZ29YMECQsiIESPoEktLy+fPn9dnADWtXbt2hJCtW7dKS/7666+ag6npRKvAWcoWLFjA4XDmz59/7Ngx2fLXr19v2bKlrKyMfjl9+nS6skAgoEvS09M3b95cbWu2traRkZFVVVWDBg3atWtXtauUV69elb69brNnz9bR0dm4caN0km6RSDRnzpzS0tIRI0a0bNmSLuzZsychJCoqqrCwkC5JTU1dunRprdt8+/ZtSEiI9P8QcXFxf/75J5fLpbue1srOzu7MmTMtWrQ4ePDg5MmT63+Nup7xV2NgYODv719ZWTlv3jyhUEgXnj59uub/hwCaDlV1V4VPQY83oCjK0dHxiy++GDNmTJcuXehpzNzd3aUDA+bMmUMIsbe39/f3nzJlyg8//ECX0x0oam42MTGRwWDQGxk/fjw9gjA0NJT8b7/5J0+esNlsBoMxYMCAyZMnT5ky5d69e5JPGD4hkUhiY2M5HA4hxMHBYfjw4X5+fh07dqRPPaUDQsrLyz///HNCiImJyejRowcMGMBkMmfOnFlzaxKJ5JdffqHfbmxsPGDAgC+//HLgwIF0pieEDBkyhO4TK6ltIJ3U6tWrCSG6urr9+vUbN24cfa7ctm3bZ8+eSeuIxWI6F5qamn7xxRfdunXT09OjD1rN4RP+/v58Pt/e3n7cuHE9e/akb8Vt37692sGp+ddJS0tr3rw5IWT27NmyG6xjHGE946/Z/JcvX9JTJVhbW48bN6537946Ojr0ccbwCWiSkAg10suXL3ft2uXv7+/i4mJkZKSrq2tiYtK7d+/NmzfLDod/+/ZtUFCQtbV1tdHQH0qEEonk8uXLffr0MTQ05PF4Xl5esbGxtSazU6dOeXt7GxgY0L/a9Ci6T0mEEonk/v3706dPb9WqFYvF4vP5Tk5OkyZNOn78uOxEASUlJaGhoS1btmQyme3atfvhhx/oq4W1DpfMz89fsmSJu7u7oaGhrq6uoaGhh4fHvHnzbty4IVutjkQokUiOHz/et29fPp/PZDLbtGmzaNGiagMBJRJJUVHRjBkzzMzMWCxW+/bto6KiarZRupe7d+/6+fkZGhpyOJyuXbseO3as5sGp9a+TnJysr69PCFmyZImkfomwPvHX2vznz59PnTrV1NSUzWa7urru2LHjQ39KgCaAkjRmb0AAoMXExEyaNGn58uUqnBUdAGqFe4QAAKDVMHwCALSaWCymJ50HlWAymdUe06Z8SIQAoL1EIlFOTk79J0AAhdPR0bG3t2/QGFmFwz1CANBSEokkPz+/oqLC0tJS5Scl2kksFj99+lRPT8/GxkaF8xnhjBAAtFRlZWVpaamlpaVip6+DBmnRosXTp08rKyvph4CqBP4TBABaih57o9qLckAf/wbNmqRwWn1GSJ+V83g8TDEM0LRJJJKSkpJaL4Hi669a6nD8tToRPn36tNq8mgDQhD1+/NjKykrVUShDTExMcHBw/adX1HJanQh5PB4h5PHjx3U89AAAmgCBQGBtbU1/5esgLn4tLn1bd52G0uEa6PCN6qjwoVOigICADz1ssiY7O7vg4ODg4GD65dixYwcNGtSgOBWoWjDqT6sTIf35a9asGRIhgDb4yFMki18XRy0nVQp7rMq/GLr8mSvqyIUFBQX0wv79+5ctWyadzp6efVc+HA7nU96ubdBZBgCAEELEpW8VnwUJIVWVdZ9lmr/H5/MpipK+vHDhAv1UrFatWq1YsUL60JLw8HAbGxsWi2VpaTl37lxCSK9evfLy8ubPn09RFJ3sY2Ji6Fna6fodO3b8/fff7ezs+Hz+uHHjSkpK6FUlJSX+/v76+voWFhYbNmzo1atXradxN2/e7N27N4/Ha9asmbu7u/R5JsnJyT169OBwONbW1nPnzn337l2twag/JEIAALVz6tSpCRMmzJ079+7du9u2bYuJiaGfJfLXX39t2LBh27Zt2dnZhw8f7tChAyHk4MGDVlZWK1euLCgokJ5fynr48OHhw4ePHz9+/PjxxMTENWvW0OUhISGXLl06evRofHx8UlJSampqrcH4+/tbWVldv349JSVlyZIl9DiHjIwMX1/fESNG3Lp1a//+/RcvXgwKCqpPMGpIqy+NAgCop9WrVy9ZsoR+UGWrVq2+++67RYsWLV++PD8/39zcvF+/fvQgdE9PT0KIkZERg8Hg8Xjm5ua1bk0sFsfExNC3SL/66quzZ8+uXr26pKRk586de/bs6du3LyEkOjpa+kzvavLz8xcuXOjo6EgIadu2LV24bt26L7/8kj6DbNu27aZNm3r27BkVFfXRYNQQzggBANROSkrKypUrDd6bNm1aQUFBaWnp6NGjy8rKWrVqNW3atEOHDkmvl9bNzs5O2lHIwsLixYsXhJBHjx5VVFTQqZQQwufzHRwcan17SEjI1KlT+/Xrt2bNmocPH0ojjImJkUbo6+srFourPS1cUyARAgCoHbFYvGLFivT3MjIysrOz2Wy2tbV1VlbW5s2bORzOrFmzevToUVFR8dGtyU7aQlEUPbcqPb+m7G28D824GR4efufOncGDB587d87Z2fnQoUN0hNOnT5dGePPmzezs7NatW39iw1UCl0YBANSOm5tbVlZWmzZtaq7icDjDhg0bNmzY7NmzHR0dMzIy3NzcmExmQydnad26tZ6e3rVr1+jh1AKBIDs7u2fPnrVWbteuXbt27ebPnz9+/Pjo6OgvvvjCzc3tzp07tUYoRzCqhUQIAKB2li1bNmTIEGtr69GjR+vo6Ny6dSsjI2PVqlUxMTFVVVWff/45l8v9/fffORyOra0tIcTOzu7ChQvjxo1jsVgmJib12QWPxwsICFi4cKGRkZGpqeny5ct1dHRq9vMsKytbuHDhqFGj7O3t//nnn+vXr48cOZIQsnjx4i5dusyePXvatGn6+vqZmZnx8fE///yzfMGoFi6NAgCoHV9f3+PHj8fHx3fu3LlLly6RkZF0wmvevPn27du9vb1dXV3Pnj177NgxY2NjQsjKlStzc3Nbt27dokWL+u8lMjKya9euQ4YM6devn7e3t5OTE5vNrlaHwWAUFhZOnDixXbt2Y8aMGThw4IoVKwghrq6uiYmJ2dnZ3bt379Sp09KlSy0sLOi3yBeMCmn1Y5gEAgGfzy8uLsaAeoCmrdYve3l5eU5Ojr29Pf3rr6oB9Wri3bt3LVu2XL9+/ZQpU5S532p/BZXApVEAAEII0eEb8WeuUP4UayqUlpZ27949T0/P4uLilStXEkL8/PxUHZQKIBECAPxLh2+ktkmrkfz4449ZWVlMJtPd3T0pKUkjbukpHBIhAICW6tSpU0pKiqqjUD0kQhK84Q2TrUk9fdVHaH4QIYRQhBDCFJdrxqyCamad9c/0gohiEw2ZmFGzsJmEECIsF6g6EFBfSIQgP5aknBBCtLe7lQKIdPCIgMZVLiKEEJFI1XGAGsPwCQAA0GpNLRFu2bKF7oZL3/hVdTgAAKDumlQi3L9/f3Bw8LfffpuWlta9e/eBAwfm5+erOigAAFBrTSoRRkZGTpkyZerUqU5OTj/99JO1tXVUVJSqgwIAALXWdBKhSCRKSUnx8fGRlvj4+CQnJ1erJhQKBTKUGyMAgOJ96MnysnJzcymKSk9PV05ImqXp9Bp99epVVVWVmZmZtMTMzOzZs2fVqkVERNAT5QEAVPP8lbBYoOAp1vjNdM1MWHVUqDnPNS0gICAmJqY+uzh48KDsg5ZqZW1tXVBQoKrx8oGBgUVFRYcPH1bJ3j+q6SRCWrVna9X8hIWFhYWEhNDLAoGAfv4IAMDzV8KJ826KKhQ8HoipR+3a+FkdubCgoIBe2L9//7Jly7KysuiXHM7/D62pqKioI9UZGX18NhwGg6FBj4xXsqZzadTExITBYMieAr548UL2BJHGYrGayVBujACgvooFlQrPgoQQUYWk7rNM8/f4fD5FUfRyeXl58+bNDxw40KtXLzab/ccffxQWFo4fP97KyorL5Xbo0GHv3r3SLcheGrWzs/v+++8nT57M4/FsbGx+/fVXulz20mhCQgJFUWfPnvXw8OByuV5eXtLsSwhZtWqVqakpj8ebOnXqkiVLOnbsWDPmN2/e+Pv7t2jRgsPhtG3bNjo6mi5/8uTJ2LFjDQ0NjY2N/fz8cnNzCSHh4eE7d+48cuQIRVEURSUkJHzSAW0ETScR0nPlxcfHS0vi4+O9vLxUGBIAwKdYvHjx3LlzMzMzfX19y8vL3d3djx8/fvv27a+//vqrr766evVqre9av369h4dHWlrarFmzZs6cee/evVqrffvtt+vXr79x44auru7kyZPpwt27d69evXrt2rUpKSk2NjYf6m+4dOnSu3fv/v3335mZmVFRUfQV19LS0t69exsYGFy4cOHixYsGBgYDBgwQiUShoaFjxowZMGBAQUFBQUGBGv4sN6lLoyEhIV999ZWHh0fXrl1//fXX/Pz8GTNmqDooAAA5BQcHjxgxQvoyNDSUXpgzZ87Jkyf//PPPzz//vOa7Bg0aNGvWLELI4sWLN2zYkJCQ4OjoWLPa6tWr6efRL1myZPDgweXl5Ww2++eff54yZcqkSZMIIcuWLTt9+vTbt7U8jiM/P79Tp04eHh6EEDs7O7pw3759Ojo6//3vf+l7UtHR0c2bN09ISPDx8eFwOEKhUG2vzTapRDh27NjCwsKVK1cWFBS4uLjExcXRj7IEANBEdKahVVVVrVmzZv/+/U+ePBEKhUKhUF9fv9Z3ubq60gv0hdYXL17UXY1+oO6LFy9sbGyysrLoJErz9PQ8d+5czffOnDlz5MiRqampPj4+w4cPp0/yUlJSHjx4wOPxpNXKy8sfPnzYoCarRJNKhISQWbNmyf4VoVEJKTYhmHT7kzDFZfQCJt1uJPSk25RY1XHIRTbVrV+/fsOGDT/99FOHDh309fWDg4NFH5hBVbZbDUVRYnHtjZdWo0/gpNWqdTms9b0DBw7My8s7ceLEmTNn+vbtO3v27B9//FEsFru7u+/evVu2pkY8pL6pJUI5/DTfEL1m5LX741WgTptUHYCWEAgYW79RdRCfJikpyc/Pb8KECYQQsVicnZ3t5OSk8L04ODhcu3btq6++ol/euHHjQzVbtGgRGBgYGBjYvXv3hQsX/vjjj25ubvv37zc1Na35i8pkMquq1PchP02nswwAQBPWpk2b+Pj45OTkzMzM6dOn1xwkrRBz5szZsWPHzp07s7OzV61adevWrVqHOS5btuzIkSMPHjy4c+fO8ePH6ZTs7+9vYmLi5+eXlJSUk5OTmJg4b968f/75hxBiZ2d369atrKysV69eVVRUNEbknwKJEABAAyxdutTNzc3X17dXr17m5ubDhw9vjL34+/uHhYWFhoa6ubnl5OQEBgay2eya1ZhMZlhYmKura48ePRgMxr59+wghXC73woULNjY2I0aMcHJymjx5cllZGX12OG3aNAcHBw8PjxYtWly6dKkxIv8U1IcuAWsDgUDA5/OLi4txaRSgaav1y15eXp6Tk0M/r4aobkC9Ouvfv7+5ufnvv//eeLuo9ldQCdwjBAAghBAzE9aujZ8pf4o1tVJaWrp161ZfX18Gg7F3794zZ87IDs5uqpAIAQD+ZWbC0qCk1RgoioqLi1u1apVQKHRwcIiNje3Xr5+qg2p0SIQAAPAvDodz5swZVUehbOgsAwAAWg2JEAAAtBoSIQBoNW3uOa8O1OH4IxECgJZiMBiEkA9NVAbKQR9/+m+hKugsAwBaSldXl8vlvnz5Uk9PT0cHZwUqIBaLX758yeVydXVVmYyQCAFAS1EUZWFhkZOTk5eXp+pYtJeOjo6NjU2tE7kpDRIhAGgvJpPZtm1bXB1VISaTqfLTcSRCANBqOjo6KpzcC9QBLosDAIBWQyIEAACthkQIAABaDfcISfCGN0y2+j46mRaaH0QIIRRhissJIfL1r1pn/bN0WUSxiUq7aUGDKOQDoCjSD5KIYhNC1PyDxGYSQoiwXKDqQEB9IRFqBpaknBBCPm0GBpEORyHBgPIp5AOgKJr1QSoXEUIIuoVCHXBpFAAAtFqTSoQXLlwYOnSopaUlRVGHDx9WdTgAAKABmlQifPfu3WefffbLL7+oOhAAANAYTeoe4cCBAwcOHKjqKAAAQJM0qURYH0KhUCgU0ssCATqSAQBouyZ1abQ+IiIi+O9ZW1urOhwAAFAxrUuEYWFhxe89fvxY1eEAAICKad2lURaLxWKxVB0FAACoC607IwQAAJDVpM4I3759++DBA3o5JycnPT3dyMjIxsZGtVEBAIA6a1KJ8MaNG71796aXQ0JCCCEBAQExMTEqDQoAANRak0qEvXr1kkjUYzZGRRP+O7vxJ825zBSXSZcx6bZmUcgHQFGkHyQNmnSbEqs6DlBjVFPNHPUhEAj4fH5xcXGzZs1UHQsANCJ82aEOcp4RFhUVXbt27cWLF2Lx//9Ha+LEiQqKCgAAQEnkSYTHjh3z9/d/9+4dj8ej3l8VoSgKiRAAADSOPMMnFixYMHny5JKSkqKiojfvvX79WuHBAQAANDZ5EuGTJ0/mzp3L5XIVHg0AAICSyZMIfX19b9y4ofBQAAAAlE+ee4SDBw9euHDh3bt3O3TooKenJy0fNmyY4gIDAABQBnmGT+jo1HIeSVFUVVWVIkJSHvSoBtAS+LJDHeQ5I5QdMgEAAKDRMOk2AABoNTkTYWJi4tChQ9u0adO2bdthw4YlJSUpNiwAAADlkCcR/vHHH/369eNyuXPnzg0KCuJwOH379t2zZ4/CgwMAAGhs8nSWcXJy+vrrr+fPny8tiYyM3L59e2ZmpkJja3S4fw6gJfBlhzrIc0b46NGjoUOHypYMGzYsJydHQSEBAAAojzyJ0Nra+uzZs7IlZ8+etba2VlBIAAAAyiPP8IkFCxbMnTs3PT3dy8uLoqiLFy/GxMRs3LhR4cEBAAA0NnkS4cyZM83NzdevX3/gwAFCiJOT0/79+/38/BQdGwAAQKPDg3lx/xyg6cOXHeog54N5m5LgDW+YbA2bHE4+oflBhBBCEUIIU1xOffIG11n/LF0WUWxCffomQQOE5gdJP0Xk3w+UMkg/byKKTQip5+eNzSSEEGG5oLHCAs3XgERoZGR0//59ExMTQ0NDqraPIB5JqOZYknJCCFHcJQCRDkdh2wLNwZKUK/BTVH/yfd7KRYQQIhIpOBhoShqQCDds2MDj8eiFWhOhakVERBw8ePDevXscDsfLy2vt2rUODjJm0Q8AAB+FSURBVA6qDgoAANRdAxJhQEAAvRAYGNg4wXySxMTE2bNnd+7cubKy8ttvv/Xx8bl7966+vr6q4wIAALUmzz1CBoNRUFBgamoqLSksLDQ1NVXtY5hOnjwpXY6OjjY1NU1JSenRo4cKQwIAAPUnz4D6mh1NhUIhk8lURDyKUVxcTAgxMjJSdSAAAKDuGnZGuGnTJkIIRVH//e9/DQwM6MKqqqoLFy44OjoqPjq5SCSSkJCQbt26ubi41FwrFAqFQiG9LBCgIxkAgLZrWCLcsGEDIUQikWzdupXBYNCFTCbTzs5u69atio9OLkFBQbdu3bp48WKtayMiIlasWKHkkAAAQG01LBHSM2v37t374MGDhoaGjRPSJ5kzZ87Ro0cvXLhgZWVVa4WwsLCQkBB6WSAQYIpUAAAtJ09nmfPnzys8jk8nkUjmzJlz6NChhIQEe3v7D1VjsVgsFkuZgQEAgDqTc2aZf/755+jRo/n5+SKZcaqRkZEKikoes2fP3rNnz5EjR3g83rNnzwghfD6fw8GIbwAAqIs8ifDs2bPDhg2zt7fPyspycXHJzc2VSCRubm4KD65BoqKiCCG9evWSlkRHR6vnkEcAAFAf8iTCsLCwBQsWrFy5ksfjxcbGmpqa+vv7DxgwQOHBNYg2zx4OAAByk2ccYWZmJj3LjK6ubllZmYGBwcqVK9euXavo2EDBhBRbSLGFOmyhDlsh/2tgisuk/wj+I6I1ZD9Fyvyr/8+Hrd6fNzaTsJmEpUbjnEHtyHNGqK+vTw/Fs7S0fPjwYfv27Qkhr169UnBoyvLTfEOteTLLbsVubpNiNwcaQ8EfpHr6lM+bQMDY+o3CIoEmRp5E2KVLl0uXLjk7Ow8ePHjBggUZGRkHDx7s0qWLwoMDAABobPIkwsjIyLdv3xJCwsPD3759u3///jZt2tBj7QEAADRLgxNhVVXV48ePXV1dCSFcLnfLli2NEBUAAICSNLizDIPB8PX1LSoqaoxoAAAAlEyeXqMdOnR49OiRwkMBAABQPnkS4erVq0NDQ48fP15QUCCQofDgAAAAGhslxzh0HZ1/0ydFUfSCRCKhKEq1D+aVg0Ag4PP5xcXFWjN8AkBL4csOdWg6k24DAADIQZ5E2LNnT4XHAQAAoBLy3CMkhCQlJU2YMMHLy+vJkyeEkN9///1DD8IFAABQZ/IkwtjYWF9fXw6Hk5qaSs+1VlJS8v333ys6NgAAgEYnTyJctWrV1q1bt2/frqenR5d4eXmlpqYqNDAAAABlkCcRZmVl9ejRQ7akWbNmGGIPAACaSJ5EaGFh8eDBA9mSixcvtmrVSkEhAQAAKI88iXD69Onz5s27evUqRVFPnz7dvXt3aGjorFmzFB4cAABAY5Nn+MSiRYuKi4t79+5dXl7eo0cPFosVGhoaFBSk8OAAAAAamzwzy9BKS0vv3r0rFoudnZ0NDAwUG5Zy0JNN3JkylMfUU3UsH7HO+md6QUSxCSHk/Zw+oOlC84MIRQghTHF5HX9UfAAIfawIIRRhissJIfU8BBSbQwgpEVY4bzmAmWWgVvJcGp08eXJJSQmXy/Xw8PD09DQwMHj37t3kyZMVHhxIiXQ49D9CUdr5I9hUsSTlLHE5q84sSPABIITQx0ry77Gq/yGQlJdJysskwrJGjAw0nDyJcOfOnWVl//OpKisr27Vrl4JCAgAAUJ6GJUKBQFBcXCyRSEpKSqQPnXjz5k1cXJypqWkjhVhPUVFRrq6uzZo1a9asWdeuXf/++2/VxgMAABqhYZ1lmjdvTlEURVHt2rWTLacoasWKFQoNrMGsrKzWrFnTpk0bQsjOnTv9/PzS0tLat2+v2qgAAEDNNSwRnj9/XiKR9OnTJzY21sjIiC5kMpm2traWlpaNEF4DDB06VLq8evXqqKioK1euIBECAEDdGpYI6edO5OTk2NjYUOp6x76qqurPP/989+5d165da64VCoX0/KiEEDxMGAAAGpAIb9265eLioqOjU1xcnJGRUbOCq6ur4gKTR0ZGRteuXcvLyw0MDA4dOuTs7FyzTkREhMqv4gIAgPpoQCLs2LHjs2fPTE1NO3bsSFHVByCqwxPqHRwc0tPTi4qKYmNjAwICEhMTa+bCsLCwkJAQelkgEFhbWys9TAAAUCMNSIQ5OTktWrSgFxotnk/CZDLpzjIeHh7Xr1/fuHHjtm3bqtVhsVgsFksV0QEAgDpqQCK0tbWttqDOJBKJ9F4gAADAh8gz16h6+uabbwYOHGhtbV1SUrJv376EhISTJ0+qOigAAFB3TScRPn/+/KuvviooKODz+a6uridPnuzfv7+qgwIAAHXXdBLhjh07VB1CI2KK/53TTpvnXG6ShBS7PpNu4wNA6GNF5Jx0m6Kazm8dKFzDnj5RWVmpq9t0Pk/00ycwIT1Ak4cvO9ShYXONWlhYhIaGZmZmNlI0AAAAStawRBgSEnLs2DEXF5euXbvu2LHj7du3jRQWAACAcjQsEYaFhWVlZSUkJDg6OgYHB1tYWEyaNOnSpUuNFBwAAEBjk+d5hN27d4+Ojn727NlPP/304MGD7t27Ozg4/PDDDwoPDgAAoLE1rLNMrU6cODFx4sSioiKVT7HWULh/DqAl8GWHOshzRkgrLS2Njo7u0aPHsGHDjI2NV69ercCwAAAAlEOesRBJSUnR0dF//fVXVVXVqFGjVq1a1aNHD4VHBgAAoAQNS4Tff/99TEzMw4cPPTw81q1bN378eFxnAAAAjdawRLhhw4YJEyZMmTLFxcWlkQICAABQpoYlwqdPn+rp6TVSKAAAAMrXsM4ySUlJzs7OAoFAtrC4uLh9+/ZJSUkKDQwAAEAZGpYIf/rpp2nTplW7L8jn86dPnx4ZGanQwAAAAJShYYnw5s2bAwYMqFnu4+OTkpKioJAAAACUp2GJ8Pnz57XeI9TV1X358qWCQgIAAFCehiXCli1bZmRk1Cy/deuWhYWFgkICAABQnoYlwkGDBi1btqy8vFy2sKysbPny5UOGDFFoYAAAAMrQsLlGnz9/7ubmxmAwgoKCHBwcKIrKzMzcvHlzVVVVamqqmZlZ4wXaGDD9IICWwJcd6tCwcYRmZmbJyckzZ84MCwujMyhFUb6+vlu2bNG4LCj1580nXAPBx+s1xIjiTfSCLhERQijFbr3JORl9hj5IlRWVqo4Fmhpdjj4hpFSkYY8EAGVq8Fyjtra2cXFxb968efDggUQiadu2raGhoQIDys3Ntbe3T0tL69ixowI3q2R6RKTqEDRJZQV+pKCxVJa9I/iMQZ3kmXSbEGJoaNi5c2fFhgIAAKB88j+GCQAAoAlQcSIUi8Vr165t06YNi8WysbGp9lDDqqqqKVOm2NvbczgcBweHjRs3SlclJCR4enrq6+s3b97c29s7Ly+PEHLz5s3evXvzeLxmzZq5u7vfuHFD2e0BAABNI+elUUUJCwvbvn37hg0bunXrVlBQcO/ePdm1YrHYysrqwIEDJiYmycnJX3/9tYWFxZgxYyorK4cPHz5t2rS9e/eKRKJr165RFEUI8ff379SpU1RUFIPBSE9Px/zgAADwUapMhCUlJRs3bvzll18CAgIIIa1bt+7WrVtubq60gp6e3ooVK+hle3v75OTkAwcOjBkzRiAQFBcXDxkypHXr1oQQJycnuk5+fv7ChQsdHR0JIW3btq11p0KhUCgU0svVZg8HAAAtpMpLo5mZmUKhsG/fvnXU2bp1q4eHR4sWLQwMDLZv356fn08IMTIyCgwM9PX1HTp06MaNGwsKCujKISEhU6dO7dev35o1ax4+fFjrBiMiIvjvWVtbK7xRAACgWVSZCDkcTt0VDhw4MH/+/MmTJ58+fTo9PX3SpEki0b/DEqKjoy9fvuzl5bV///527dpduXKFEBIeHn7nzp3BgwefO3fO2dn50KFDNbcZFhZW/N7jx48V3igAANAsqkyEbdu25XA4Z8+e/VCFpKQkLy+vWbNmderUqU2bNtVO8jp16hQWFpacnOzi4rJnzx66sF27dvPnzz99+vSIESOio6NrbpPFYjWTodgWAQCAxlFlImSz2YsXL160aNGuXbsePnx45cqVHTt2yFZo06bNjRs3Tp06df/+/aVLl16/fp0uz8nJCQsLu3z5cl5e3unTp+/fv+/k5FRWVhYUFJSQkJCXl3fp0qXr169L7x0CAAB8iIp7jS5dulRXV3fZsmVPnz61sLCYMWOG7NoZM2akp6ePHTuWoqjx48fPmjXr77//JoRwudx79+7t3LmzsLDQwsIiKCho+vTplZWVhYWFEydOfP78uYmJyYgRI6QdbQAAAD6kYZNuNzH0PLz/vXCXa8BT7JbHFv+o2A02bcd/PaXqEKCJK62oGv9XNibdhlqp+IywqaogTHoBk27Xh64eA5NuQyOhJ93WZWCuUfggnBHiySwATR++7FAHzDUKAABaDYkQAAC0GhIhAABoNSRCAADQakiEAACg1ZAIAQBAqyERAgCAVkMiBAAArYZECAAAWg2JEAAAtBoSIQAAaDUkQgAA0GpIhAAAoNWQCAEAQKshEQIAgFZDIgQAAK2GRAgAAFpNV9UBqN6fN59wDQTK3GP4wQzpsrBSXJ+3/DDpHiGEIoSlJyaEUJQ8++1/7v+fzc2oIhSRaysynn+/lryPRiIUfuLWPsr8v8MIvT82gxBCyXcU1ExaySR6QUz0yIf/Irs3XqYXKkRVygirTj/+LaEDZXEJkffTKIe4KZvphcoyUf3fpcvRJ4SUqsFxA7WFRKgC9Ux+sjjMBr+lJt0qBf9iSUQN+D36dDocPWXuTjnEhFmfauqQ/6Q4+qrZb4Pyn8y73hFCKivU6ACCusGlUQAA0GpqlwhFyj3JAAAALafURFhSUuLv76+vr29hYbFhw4ZevXoFBwcTQuzs7FatWhUYGMjn86dNm0YIiY2Nbd++PYvFsrOzW79+vXQLFEUdPnxY+rJ58+YxMTGEkNzcXIqi9u3b5+XlxWaz27dvn5CQoMymAQCAhlJqIgwJCbl06dLRo0fj4+OTkpJSU1Olq9atW+fi4pKSkrJ06dKUlJQxY8aMGzcuIyMjPDx86dKldLb7qIULFy5YsCAtLc3Ly2vYsGGFhYWN1hQAAGgilNdZpqSkZOfOnXv27Onbty8hJDo62tLSUrq2T58+oaGh9LK/v3/fvn2XLl1KCGnXrt3du3fXrVsXGBj40V0EBQWNHDmSEBIVFXXy5MkdO3YsWrSoWh2hUCh8379RIFBqZ1EAAFBDyjsjfPToUUVFhaenJ/2Sz+c7ODhI13p4eEiXMzMzvb29pS+9vb2zs7Orqj7e6atr1670gq6uroeHR2ZmZs06ERER/Pesra3lawsAADQZykuEEomE/O/YL7qEpq+vL1v+oWoURcm+rKioqGOPtY4zCwsLK37v8ePHDWsDAAA0OcpLhK1bt9bT07t27Rr9UiAQZGdn11rT2dn54sWL0pfJycnt2rVjMBiEkBYtWhQUFNDl2dnZpaWlsm+8cuUKvVBZWZmSkuLo6Fhz4ywWq5mMT24WAABoNuXdI+TxeAEBAQsXLjQyMjI1NV2+fLmOjk6tJ20LFizo3Lnzd999N3bs2MuXL//yyy9btmyhV/Xp0+eXX37p0qWLWCxevHixnt7/jLDevHlz27ZtnZycNmzY8ObNm8mTJyujYQAAoMmU2ms0MjKya9euQ4YM6devn7e3t5OTE5vNrlnNzc3twIED+/btc3FxWbZs2cqVK6U9ZdavX29tbd2jR48vv/wyNDSUy+XKvnHNmjVr16797LPPkpKSjhw5YmJiooxWAQCAJlPqFGs8Hm/37t308rt371asWPH1118TQnJzc6vVHDlyJN3/sxpLS8tTp05JXxYVFcmudXJykl4dBQAAqA+lJsK0tLR79+55enoWFxevXLmSEOLn56fMANQES/f/T8TrOe9omUiHfPKk25WM/+9npJBJtykmkxDlTbotLqP7RjWpSbd1yL9TKdU96bYek0EvqMOko2XviEom3dbl/DsvqxyTbusyVH/cQG39TyfMxpaWljZ16tSsrCwmk+nu7h4ZGdmhQweFbDk3N9fe3j4tLa1jx471f5dAIODz+cXFxeg1A9C04csOdVBqIlQ3+G4AaAl82aEOajfpNgAAgDIhEQIAgFZDIgQAAK2GRAgAAFoNiRAAALQaEiEAAGg1JEIAANBqSIQAAKDVkAgBAECrIRECAIBWQyIEAACthkQIAABaTamPYVI39ITjAoFA1YEAQOOiv+ba/IwBqINWJ8LCwkJCiLW1taoDAQBlKCws5PP5qo4C1I5WJ0IjIyNCSH5+vqZ/NwQCgbW19ePHjzX9ETNoiLppMg0pLi62sbGhv/IA1Wh1ItTR0SGE8Pl8Tf+S05o1a4aGqBU0RN3QX3mAavCxAAAArYZECAAAWo0RHh6u6hhUicFg9OrVS1dX4y8RoyHqBg1RN02mIaBwFPoTAwCANsOlUQAA0GpIhAAAoNWQCAEAQKtpbyLcsmWLvb09m812d3dPSkpSdTj1FRER0blzZx6PZ2pqOnz48KysLOkqiUQSHh5uaWnJ4XB69ep1584dFcbZIBERERRFBQcH0y81riFPnjyZMGGCsbExl8vt2LFjSkoKXa5ZDamsrPzPf/5jb2/P4XBatWq1cuVKsVhMr1L/hly4cGHo0KGWlpYURR0+fFhaXkfkQqFwzpw5JiYm+vr6w4YN++eff1QROKgHiVbat2+fnp7e9u3b7969O2/ePH19/by8PFUHVS++vr7R0dG3b99OT08fPHiwjY3N27dv6VVr1qzh8XixsbEZGRljx461sLAQCASqjbY+rl27Zmdn5+rqOm/ePLpEsxry+vVrW1vbwMDAq1ev5uTknDlz5sGDB/QqzWrIqlWrjI2Njx8/npOT8+effxoYGPz000/0KvVvSFxc3LfffhsbG0sIOXTokLS8jshnzJjRsmXL+Pj41NTU3r17f/bZZ5WVlSoKH1RMSxOhp6fnjBkzpC8dHR2XLFmiwnjk8+LFC0JIYmKiRCIRi8Xm5uZr1qyhV5WXl/P5/K1bt6o0wI8rKSlp27ZtfHx8z5496USocQ1ZvHhxt27dapZrXEMGDx48efJk6csRI0ZMmDBBomkNkU2EdUReVFSkp6e3b98+etWTJ090dHROnjypkphB5bTx0qhIJEpJSfHx8ZGW+Pj4JCcnqzAk+RQXF5P3M6bm5OQ8e/ZM2igWi9WzZ0/1b9Ts2bMHDx7cr18/aYnGNeTo0aMeHh6jR482NTXt1KnT9u3b6XKNa0i3bt3Onj17//59QsjNmzcvXrw4aNAgooENkaoj8pSUlIqKCukqS0tLFxcXjWgUNAZtHFv66tWrqqoqMzMzaYmZmdmzZ89UGJIcJBJJSEhIt27dXFxcCCF0/NUalZeXp7L46mHfvn2pqanXr1+XLdS4hjx69CgqKiokJOSbb765du3a3LlzWSzWxIkTNa4hixcvLi4udnR0ZDAYVVVVq1evHj9+PNHAv4hUHZE/e/aMyWQaGhrKrtK4HwFQFG1MhDSKoqTLEolE9qVGCAoKunXr1sWLF2ULNahRjx8/njdv3unTp9lsds21GtQQsVjs4eHx/fffE0I6dep0586dqKioiRMn0ms1qCH79+//448/9uzZ0759+/T09ODgYEtLy4CAAHqtBjWkmnpGrlmNAsXSxkujJiYmDAZD9n9/L168kP1vo/qbM2fO0aNHz58/b2VlRZeYm5uT9/8Fpql5o1JSUl68eOHu7q6rq6urq5uYmLhp0yZdXV06Zg1qiIWFhbOzs/Slk5NTfn4+0cC/yMKFC5csWTJu3LgOHTp89dVX8+fPj4iIIBrYEKk6Ijc3NxeJRG/evKm5CrSQNiZCJpPp7u4eHx8vLYmPj/fy8lJhSPUnkUiCgoIOHjx47tw5e3t7abm9vb25ubm0USKRKDExUZ0b1bdv34yMjPT3PDw8/P3909PTW7VqpVkN8fb2lh3Ecv/+fVtbW6KBf5HS0lLZpxQxGAx6+ITGNUSqjsjd3d319PSkqwoKCm7fvq0RjYJGobJuOipFD5/YsWPH3bt3g4OD9fX1c3NzVR1UvcycOZPP5yckJBS8V1paSq9as2YNn88/ePBgRkbG+PHj1bCPex2kvUYlmtaQa9eu6erqrl69Ojs7e/fu3Vwu948//qBXaVZDAgICWrZsSQ+fOHjwoImJyaJFi+hV6t+QkpKStLS0tLQ0QkhkZGRaWho9IKqOyGfMmGFlZXXmzJnU1NQ+ffpg+IQ209JEKJFINm/ebGtry2Qy3dzc6BEIGqHmf2Wio6PpVWKxePny5ebm5iwWq0ePHhkZGaoNtUFkE6HGNeTYsWMuLi4sFsvR0fHXX3+VlmtWQwQCwbx582xsbNhsdqtWrb799luhUEivUv+GnD9/vtr3IiAgQFJn5GVlZUFBQUZGRhwOZ8iQIfn5+aoLH1QMT58AAACtpo33CAEAAKSQCAEAQKshEQIAgFZDIgQAAK2GRAgAAFoNiRAAALQaEiEAAGg1JEIAANBqSITQxFEUdfjw4frXT0hIoCiqqKioQXsJDAwcPnx4A0MDALWARAhK8uLFi+nTp9vY2LBYLHNzc19f38uXL6s6qFp4eXkVFBTw+XxVBwIASqK9zyMEJRs5cmRFRcXOnTtbtWr1/Pnzs2fPvn79WtVB1YLJZNKP7wEALYEzQlCGoqKiixcvrl27tnfv3ra2tp6enmFhYYMHD6bXRkZGdujQQV9f39raetasWW/fvqXLY2Jimjdvfvz4cQcHBy6XO2rUqHfv3u3cudPOzs7Q0HDOnDlVVVV0TTs7u+++++7LL780MDCwtLT8+eefaw3jyZMnY8eONTQ0NDY29vPzy83NrVlH9tIoHcCpU6ecnJwMDAwGDBhQUFBAV6uqqgoJCWnevLmxsTH9lAbpFiQSyQ8//NCqVSsOh/PZZ5/99ddfdGG/fv0GDBhA1ywqKrKxsfn2228VcngB4FMgEYIyGBgYGBgYHD58WCgU1lyro6OzadOm27dv79y589y5c4sWLZKuKi0t3bRp0759+06ePJmQkDBixIi4uLi4uLjff//9119/pXMMbd26da6urqmpqWFhYfPnz5d93qR0U7179zYwMLhw4cLFixfpxCYSieqOvLS09Mcff/z9998vXLiQn58fGhpKl69fv/63337bsWPHxYsXX79+fejQIelb/vOf/0RHR0dFRd25c2f+/PkTJkxITEykKGrnzp3Xrl3btGkTIWTGjBlmZmbh4eENPJAA0AhU+uwL0CJ//fWXoaEhm8328vIKCwu7efNmrdUOHDhgbGxML0dHRxNCHjx4QL+cPn06l8stKSmhX/r6+k6fPp1etrW1pU+2aGPHjh04cCC9TAg5dOiQRCLZsWOHg4ODWCymy4VCIYfDOXXqVLUA6Af6vHnzpmYAmzdvNjMzo5ctLCzWrFlDL1dUVFhZWfn5+Ukkkrdv37LZ7OTkZOkGp0yZMn78eGnrWCxWWFgYl8vNysqq98EDgEaEM0JQkpEjRz59+vTo0aO+vr4JCQlubm4xMTH0qvPnz/fv379ly5Y8Hm/ixImFhYXv3r2jV3G53NatW9PLZmZmdnZ2BgYG0pcvXryQbr9r166yy5mZmdUCSElJefDgAY/Ho09PjYyMysvLHz58WHfYsgFYWFjQeywuLi4oKJDuUVdX18PDg16+e/dueXl5//79Dd7btWuXdC+jR48eMWJERETE+vXr27VrV/+jBwCNB51lQHnYbHb//v379++/bNmyqVOnLl++PDAwMC8vb9CgQTNmzPjuu++MjIwuXrw4ZcqUiooK+i16enrSt1MUVe2lWCz+0L4oiqpWIhaL3d3dd+/eLVvYokWLumOutkfJx57fSYd04sSJli1bSgtZLBa9UFpampKSwmAwsrOz694OACgNEiGohrOzMz2878aNG5WVlevXr9fR0SGEHDhwQL4NXrlyRXbZ0dGxWgU3N7f9+/ebmpo2a9ZM3qj/xefzLSwsrly50qNHD0JIZWVlSkqKm5sbIcTZ2ZnFYuXn5/fs2bPmGxcsWKCjo/P3338PGjRo8ODBffr0+cRIAODTIRGCMhQWFo4ePXry5Mmurq48Hu/GjRs//PCDn58fIaR169aVlZU///zz0KFDL126tHXrVvl2cenSpR9++GH48OHx8fF//vnniRMnqlXw9/dft26dn5/fypUrrays8vPzDx48uHDhQisrKzl2N2/evDVr1rRt29bJySkyMlI6AJ/H44WGhs6fP18sFnfr1k0gECQnJxsYGAQEBJw4ceK33367fPmym5vbkiVLAgICbt26ZWhoKF97AUBRcI8QlMHAwODzzz/fsGFDjx49XFxcli5dOm3atF9++YUQ0rFjx8jIyLVr17q4uOzevTsiIkK+XSxYsCAlJaVTp07ffffd+vXrfX19q1XgcrkXLlywsbEZMWKEk5PT5MmTy8rK5D47XLBgwcSJEwMDA7t27crj8b744gvpqu+++27ZsmURERFOTk6+vr7Hjh2zt7d/+fLllClTwsPD6RPH5cuXW1pazpgxQ769A4ACffyeB4D6s7OzCw4ODg4OVnUgAKB5cEYIAABaDYkQAAC0Gi6NAgCAVsMZIQAAaDUkQgAA0GpIhAAAoNWQCAEAQKshEQIAgFZDIgQAAK2GRAgAAFoNiRAAALQaEiEAAGg1JEIAANBqSIQAAKDVkAgBAECrIRECAIBWQyIEAACt9n/7QgXrUSIWqQAAAABJRU5ErkJggg=="}}},{"cell_type":"code","source":"# gkf = KFold(n_splits=CFG.n_fold, shuffle=True, random_state=CFG.seed)\n# gkf = GroupKFold(n_splits=CFG.n_fold, shuffle=True, random_state=CFG.seed)\ngkf = StratifiedKFold(n_splits=CFG.n_fold, shuffle=True, random_state=CFG.seed)\n# gkf = StratifiedGroupKFold(n_splits=CFG.n_fold, shuffle=True, random_state=CFG.seed)\n\nfor fold, (train_id, val_id) in enumerate(gkf.split(X=df, y=df.discourse_effectiveness, groups=df.essay_id)):\n    # For all row in val_id list => create kfold column value\n    df.loc[val_id , \"kfold\"] = fold","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.697870Z","iopub.execute_input":"2022-08-02T18:46:49.698775Z","iopub.status.idle":"2022-08-02T18:46:49.723426Z","shell.execute_reply.started":"2022-08-02T18:46:49.698722Z","shell.execute_reply":"2022-08-02T18:46:49.722159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('kfold')['discourse_effectiveness'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.727926Z","iopub.execute_input":"2022-08-02T18:46:49.729583Z","iopub.status.idle":"2022-08-02T18:46:49.749114Z","shell.execute_reply.started":"2022-08-02T18:46:49.729538Z","shell.execute_reply":"2022-08-02T18:46:49.747970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', 50)\npd.set_option('display.max_columns', 20)\ndf.groupby('kfold')['discourse_type'].value_counts()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T18:46:49.751993Z","iopub.execute_input":"2022-08-02T18:46:49.752805Z","iopub.status.idle":"2022-08-02T18:46:49.771598Z","shell.execute_reply.started":"2022-08-02T18:46:49.752761Z","shell.execute_reply":"2022-08-02T18:46:49.770465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset Class","metadata":{}},{"cell_type":"code","source":"class FeedbackDataset(Dataset):\n    def __init__(self,df, max_length, tokenizer, training=True):\n        self.df = df\n        self.max_len = max_length\n        self.tokenizer = tokenizer\n        self.discourse_type = self.df['discourse_type'].values\n        self.discourse_text = self.df['discourse_text'].values\n        self.essays = self.df['essay_text'].values\n        self.training = training\n        \n        if self.training:\n            self.targets = self.df['discourse_effectiveness'].values\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        discourse_type = self.discourse_type[index]\n        discourse_text = self.discourse_text[index]\n        essay = self.essays[index]\n#         type_text = discourse_type + ' ' + discourse_text\n        \n        inputs = self.tokenizer.encode_plus(\n            discourse_text,\n            essay,\n            truncation = True,\n            add_special_tokens = True,\n            return_token_type_ids = True,\n            max_length = self.max_len\n        )\n        \n        samples = {\n            'input_ids': inputs['input_ids'],\n            'attention_mask': inputs['attention_mask'],\n        }\n        \n        if 'token_type_ids' in inputs:\n            samples['token_type_ids'] = inputs['token_type_ids']\n          \n        if self.training:\n            samples['target'] = self.targets[index]\n            samples['discourse_type'] = discourse_type\n        \n        return samples","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.772969Z","iopub.execute_input":"2022-08-02T18:46:49.773795Z","iopub.status.idle":"2022-08-02T18:46:49.785292Z","shell.execute_reply.started":"2022-08-02T18:46:49.773756Z","shell.execute_reply":"2022-08-02T18:46:49.783987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.tokenizer.encode_plus(\n            \"Hello [SEP] world\",\n            \"Hello world\",\n            truncation = True,\n            add_special_tokens = True, \n            return_token_type_ids = True,\n            max_length = 200\n        )","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.787133Z","iopub.execute_input":"2022-08-02T18:46:49.787611Z","iopub.status.idle":"2022-08-02T18:46:49.801959Z","shell.execute_reply.started":"2022-08-02T18:46:49.787576Z","shell.execute_reply":"2022-08-02T18:46:49.800835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Collator - Dynamic padding\n\nRef: https://www.kaggle.com/code/rolianklay/ensemble-deberta-v3-base-roberta-deberta-large","metadata":{}},{"cell_type":"code","source":"# Dynamic Padding (Collate)\nclass Collate:\n    def __init__(self, tokenizer, isTrain=True):\n        self.tokenizer = tokenizer\n        self.isTrain = isTrain\n        # self.args = args\n\n    def __call__(self, batch):\n        output = dict()\n        output[\"input_ids\"] = [sample[\"input_ids\"] for sample in batch]\n        output[\"attention_mask\"] = [sample[\"attention_mask\"] for sample in batch]\n        if self.isTrain:\n            output[\"target\"] = [sample[\"target\"] for sample in batch]\n            output[\"discourse_type\"] = [sample[\"discourse_type\"] for sample in batch]\n\n        # calculate max token length of this batch\n        batch_max = max([len(ids) for ids in output[\"input_ids\"]])\n\n        # add padding\n        if self.tokenizer.padding_side == \"right\":\n            output[\"input_ids\"] = [s + (batch_max - len(s)) * [self.tokenizer.pad_token_id] for s in output[\"input_ids\"]]\n            output[\"attention_mask\"] = [s + (batch_max - len(s)) * [0] for s in output[\"attention_mask\"]]\n        else:\n            output[\"input_ids\"] = [(batch_max - len(s)) * [self.tokenizer.pad_token_id] + s for s in output[\"input_ids\"]]\n            output[\"attention_mask\"] = [(batch_max - len(s)) * [0] + s for s in output[\"attention_mask\"]]\n\n        # convert to tensors\n        output[\"input_ids\"] = torch.tensor(output[\"input_ids\"], dtype=torch.long)\n        output[\"attention_mask\"] = torch.tensor(output[\"attention_mask\"], dtype=torch.long)\n        if self.isTrain:\n            output[\"target\"] = torch.tensor(output[\"target\"], dtype=torch.long)\n            output[\"discourse_type\"] = torch.tensor(output[\"discourse_type\"], dtype=torch.long)\n\n        return output\n\n# collate_fn = DataCollatorWithPadding(tokenizer=CFG.tokenizer)\ncollate_fn = Collate(CFG.tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.804495Z","iopub.execute_input":"2022-08-02T18:46:49.805348Z","iopub.status.idle":"2022-08-02T18:46:49.819053Z","shell.execute_reply.started":"2022-08-02T18:46:49.805298Z","shell.execute_reply":"2022-08-02T18:46:49.817826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building Model\n","metadata":{}},{"cell_type":"markdown","source":"## Pooling Layer","metadata":{}},{"cell_type":"code","source":"class MeanPooling(nn.Module):\n    def __init__(self):\n        super(MeanPooling, self).__init__()\n        \n    def forward(self, last_hidden_state, attention_mask):\n        input_mask_expanded = attention_mask.unsqueeze(-1).expand(last_hidden_state.size()).float()\n        sum_embeddings = torch.sum(last_hidden_state * input_mask_expanded, 1)\n        sum_mask = input_mask_expanded.sum(1)\n        sum_mask = torch.clamp(sum_mask, min=1e-9)\n        mean_embeddings = sum_embeddings / sum_mask\n\n        return mean_embeddings\n\n    \nclass MeanMaxPooling(nn.Module):\n    def __init__(self):\n        super(MeanMaxPooling, self).__init__()\n        \n    def forward(self, last_hidden_state, attention_mask):\n        mean_pooling_embeddings = torch.mean(last_hidden_state, 1)\n        _, max_pooling_embeddings = torch.max(last_hidden_state, 1)\n        mean_max_embeddings = torch.cat((mean_pooling_embeddings, max_pooling_embeddings), 1)\n        return mean_max_embeddings\n\n    \nclass LSTMPooling(nn.Module):\n    def __init__(self, num_layers, hidden_size, hiddendim_lstm):\n        super(LSTMPooling, self).__init__()\n        self.num_hidden_layers = num_layers\n        self.hidden_size = hidden_size\n        self.hiddendim_lstm = hiddendim_lstm\n        self.lstm = nn.LSTM(self.hidden_size, self.hiddendim_lstm, batch_first=True)\n        self.dropout = nn.Dropout(0.1)\n    \n    def forward(self, all_hidden_states):\n        ## forward\n        hidden_states = torch.stack([all_hidden_states[layer_i][:, 0].squeeze()\n                                     for layer_i in range(1, self.num_hidden_layers+1)], dim=-1)\n        hidden_states = hidden_states.view(-1, self.num_hidden_layers, self.hidden_size)\n        out, _ = self.lstm(hidden_states, None)\n        out = self.dropout(out[:, -1, :])\n        return out\n    \nclass WeightedLayerPooling(nn.Module):\n    def __init__(self, num_hidden_layers, layer_start: int = 4, layer_weights = None):\n        super(WeightedLayerPooling, self).__init__()\n        self.layer_start = layer_start\n        self.num_hidden_layers = num_hidden_layers\n        self.layer_weights = layer_weights if layer_weights is not None \\\n            else nn.Parameter(\n                torch.tensor([1] * (num_hidden_layers+1 - layer_start), dtype=torch.float)\n            )\n\n    def forward(self, all_hidden_states):\n        all_layer_embedding = torch.stack(list(all_hidden_states), dim=0)\n        all_layer_embedding = all_layer_embedding[self.layer_start:, :, :, :]\n        weight_factor = self.layer_weights.unsqueeze(-1).unsqueeze(-1).unsqueeze(-1).expand(all_layer_embedding.size())\n        weighted_average = (weight_factor*all_layer_embedding).sum(dim=0) / self.layer_weights.sum()\n        return weighted_average","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.823072Z","iopub.execute_input":"2022-08-02T18:46:49.823509Z","iopub.status.idle":"2022-08-02T18:46:49.842881Z","shell.execute_reply.started":"2022-08-02T18:46:49.823477Z","shell.execute_reply":"2022-08-02T18:46:49.841489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multi Sample Dropout","metadata":{}},{"cell_type":"code","source":"class MultiSampleDropout(nn.Module):\n    # Multisample Dropout: https://arxiv.org/abs/1905.09788\n    def __init__(self, classifier, start_prob=0.2, num_samples=8, increment=0.01):\n        super(MultiSampleDropout, self).__init__()\n        #self.dropout = nn.Dropout\n        self.dropouts = [StableDropout(start_prob + (increment*i)) for i in range(num_samples)] \n        self.classifier = classifier\n        \n    def forward(self, out):\n        return torch.mean(torch.stack([\n            self.classifier(dropout(out)) for dropout in self.dropouts\n        ], dim=0), dim=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.852145Z","iopub.execute_input":"2022-08-02T18:46:49.854582Z","iopub.status.idle":"2022-08-02T18:46:49.863319Z","shell.execute_reply.started":"2022-08-02T18:46:49.854546Z","shell.execute_reply":"2022-08-02T18:46:49.862082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create FeedbackModel","metadata":{}},{"cell_type":"code","source":"class FeedbackModel(nn.Module):\n    def __init__(self, model_name):\n        super(FeedbackModel, self).__init__()\n        \n        # DeBERTa\n        self.model = AutoModel.from_pretrained(model_name)\n        self.config = AutoConfig.from_pretrained(model_name)\n        \n        # gradient checkpointing\n        if CFG.gradient_checkpoint:\n            self.model.gradient_checkpointing_enable()\n            print(f\"Gradient Checkpointing: {self.model.is_gradient_checkpointing}\")\n\n        # freezing embeddings and first 6 layers of encoder\n        if  CFG.freezing:\n            freeze(self.model.embeddings)\n            freeze(self.model.encoder.layer[:6])\n            \n        # Pooling\n        #self.weighted_pooler = WeightedLayerPooling(num_hidden_layers=self.config.num_hidden_layers, layer_start=4)\n        #self.pooler = MeanPooling()\n        \n        self.context_pooler = ContextPooler(self.config)\n        \n        #self.bilstm = nn.LSTM(self.config.hidden_size, self.config.hidden_size//2, num_layers=2, \n        #                      dropout=self.config.hidden_dropout_prob, batch_first=True,\n        #                      bidirectional=False)\n        \n        #self.drop = nn.Dropout(p=0.2)\n        \n        # Multi Sample Dropout\n        self.fc = nn.Linear(self.config.hidden_size, CFG.num_classes)\n        self.multi_sample_dropout = MultiSampleDropout(self.fc, start_prob=CFG.dropout, num_samples=8, increment=0.01)\n\n        self.fc_type = nn.Linear(self.config.hidden_size, 7)\n        \n    def forward(self, ids, mask):        \n        out = self.model(input_ids=ids,attention_mask=mask,\n                        output_hidden_states=True)\n        \n        # out = self.weighted_pooler(out.hidden_states) # For WeightedLayerPooling\n        # out = self.pooler(out, mask) # For MeanPooling\n                \n        #out = self.context_pooler(torch.stack(list(out.hidden_states), dim=0)) # For ContextPooler\n        out = self.context_pooler(out[0]) # For ContextPooler\n\n        # out = self.pooler(out.last_hidden_state, mask)\n\n        outputs = self.multi_sample_dropout(out)\n        \n        #out = self.pooler(out.last_hidden_state, mask)\n        #out = self.bilstm(out)[0]\n        #out = self.drop(out)\n        #outputs = self.fc(out)\n\n        # discourse type \n        output_type = self.fc_type(out)\n        \n        return outputs, output_type\n    \n    def set_optimizer_scheduler(self, option=\"Adam8bit\"):\n        if option == \"AdamW\":\n            model_parameters = filter(lambda parameter: parameter.requires_grad, self.parameters())\n\n            # Optimizer and scheduler\n            optimizer = AdamW(model_parameters, lr=CFG.learning_rate, weight_decay = CFG.weight_decay)\n            scheduler = fetch_scheduler(optimizer)\n        elif option == \"Adam8bit\":\n            # Adam 8-bits optimizer\n            no_decay = [\"bias\", \"LayerNorm.weight\"]\n            optimizer_grouped_parameters = [\n                {\n                    \"params\": [p for n, p in self.named_parameters() if not any(nd in n for nd in no_decay) and p[1].requires_grad],\n                    \"weight_decay\": CFG.weight_decay,\n                },\n                {\n                    \"params\": [p for n, p in self.named_parameters() if any(nd in n for nd in no_decay) and p[1].requires_grad],\n                    \"weight_decay\": 0.0,\n                },\n            ]\n\n            # initializing optimizer \n            # bnb_optimizer = bnb.optim.AdamW(params=model_parameters, lr=CFG.learning_rate, weight_decay=CFG.weight_decay, optim_bits=8)\n            optimizer = bnb.optim.Adam8bit(optimizer_grouped_parameters, lr=CFG.learning_rate)\n            print(f\"8-bit Optimizer:\\n\\n{optimizer}\")\n\n            # setting embeddings parameters\n            # set_embedding_parameters_bits(embeddings_path=self.model.embeddings)\n\n            scheduler = fetch_scheduler(optimizer)\n        else:\n            embedding_parameters = filter(lambda parameter: parameter.requires_grad, self.model.parameters())\n\n            optimizer_model = AdamW(embedding_parameters, lr=5e-6, weight_decay = CFG.weight_decay)\n            optimizer_linear = AdamW(model.fc.parameters(), lr=1e-4, weight_decay = CFG.weight_decay)\n\n            scheduler_model = fetch_scheduler(optimizer_model)\n            scheduler_linear = fetch_scheduler(optimizer_linear)\n\n            optimizer = [optimizer_model, optimizer_linear]\n            scheduler = [scheduler_model, scheduler_linear]\n        \n        self.optimizer = optimizer\n        self.scheduler = scheduler\n        return True\n    \n    def get_optimizer_scheduler(self):\n        return self.optimizer, self.scheduler\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.867178Z","iopub.execute_input":"2022-08-02T18:46:49.868012Z","iopub.status.idle":"2022-08-02T18:46:49.887823Z","shell.execute_reply.started":"2022-08-02T18:46:49.867984Z","shell.execute_reply":"2022-08-02T18:46:49.886348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loss Function","metadata":{}},{"cell_type":"code","source":"class LabelSmoothing(nn.Module):\n    \"Implement label smoothing.\"\n    def __init__(self, size, padding_idx, smoothing=0.0):\n        super(LabelSmoothing, self).__init__()\n        self.criterion = nn.KLDivLoss(reduction='sum')\n        self.padding_idx = padding_idx\n        self.confidence = 1.0 - smoothing\n        self.smoothing = smoothing\n        self.size = size\n        self.true_dist = None\n        \n    def forward(self, x, target):\n        assert x.size(1) == self.size\n        true_dist = x.data.clone()\n        true_dist.fill_(self.smoothing / (self.size - 2))\n        true_dist.scatter_(1, target.data.unsqueeze(1), self.confidence)\n        true_dist[:, self.padding_idx] = 0\n        mask = torch.nonzero(target.data == self.padding_idx)\n        if mask.dim() > 0:\n            true_dist.index_fill_(0, mask.squeeze(), 0.0)\n        self.true_dist = true_dist\n        return self.criterion(x, Variable(true_dist, requires_grad=False))\n    \n# criterion = LabelSmoothing(size=3, padding_idx=CFG.tokenizer.pad_token_id, smoothing=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.890760Z","iopub.execute_input":"2022-08-02T18:46:49.891528Z","iopub.status.idle":"2022-08-02T18:46:49.903265Z","shell.execute_reply.started":"2022-08-02T18:46:49.891399Z","shell.execute_reply":"2022-08-02T18:46:49.902180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example of label smoothing.\ncrit = LabelSmoothing(5, 0, 0.4)\npredict = torch.FloatTensor([[0, 0.2, 0.7, 0.1, 0],\n                             [0, 0.2, 0.7, 0.1, 0], \n                             [0, 0.2, 0.7, 0.1, 0]])\nv = crit(Variable(predict.log()), \n         Variable(torch.LongTensor([2, 1, 0])))\n\n# Show the target distributions expected by the system.\nv.item()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.906272Z","iopub.execute_input":"2022-08-02T18:46:49.908582Z","iopub.status.idle":"2022-08-02T18:46:49.960678Z","shell.execute_reply.started":"2022-08-02T18:46:49.908545Z","shell.execute_reply":"2022-08-02T18:46:49.959511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def criterion(outputs, labels):\n    return nn.CrossEntropyLoss()(outputs, labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:46:49.962428Z","iopub.execute_input":"2022-08-02T18:46:49.963135Z","iopub.status.idle":"2022-08-02T18:46:49.970848Z","shell.execute_reply.started":"2022-08-02T18:46:49.963089Z","shell.execute_reply":"2022-08-02T18:46:49.969480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Function","metadata":{}},{"cell_type":"code","source":"def train_one_epoch(model, dataloader, device, epoch):\n    model.train()\n    dataset_size = 0\n    running_loss= 0\n    type_running_loss = 0\n    effect_running_loss = 0\n    epoch_loss=0\n    \n    optimizer, scheduler = model.get_optimizer_scheduler()\n    \n    bar = tqdm(enumerate(dataloader), total= len(dataloader))\n    for step, data in bar:\n        ids = data['input_ids'].to(device, dtype = torch.long)\n        mask = data['attention_mask'].to(device, dtype = torch.long)\n        \n        targets = data['target'].to(device, dtype = torch.long)\n        discourse_types = data['discourse_type'].to(device, dtype = torch.long)\n\n        batch_size = ids.size(0)\n        outputs, outputs_type = model(ids, mask)\n        \n        loss = criterion(outputs, targets)\n        loss_type = criterion(outputs_type, discourse_types)\n        \n        loss_combine = 3*loss/5 + 2*loss_type/5\n        loss_combine = loss_combine/CFG.n_accumulate\n        loss_combine.backward()\n       \n        if (step+1)% CFG.n_accumulate ==0:\n            optimizer.step()\n            optimizer.zero_grad()\n            scheduler.step()\n                \n        running_loss += (loss_combine.item()*batch_size) * CFG.n_accumulate\n        type_running_loss += (loss_type.item()*batch_size)\n        effect_running_loss += (loss.item()*batch_size)\n        \n        dataset_size += batch_size\n        \n        epoch_loss = running_loss / dataset_size\n        type_epoch_loss = type_running_loss / dataset_size\n        effect_epoch_loss = effect_running_loss / dataset_size\n        \n        wandb.log({'Train Combine Loss': epoch_loss})\n        wandb.log({'Train Type Loss': type_epoch_loss})\n        wandb.log({'Train Effect Loss': effect_epoch_loss})\n        \n        wandb.log({'Train Type Loss1': loss_type.item()})\n        wandb.log({'Train Effect Loss1': loss.item()})\n\n        bar.set_postfix(Epoch = epoch, Train_loss = epoch_loss, Effect_loss = loss.item(), Type_loss = loss_type.item(), LR=optimizer.param_groups[0]['lr'])\n        # bar.set_postfix(Epoch = epoch, Train_loss = epoch_loss)\n    gc.collect()\n    return epoch_loss\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:17.990394Z","iopub.execute_input":"2022-08-02T18:54:17.991193Z","iopub.status.idle":"2022-08-02T18:54:18.006966Z","shell.execute_reply.started":"2022-08-02T18:54:17.991157Z","shell.execute_reply":"2022-08-02T18:54:18.006067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation Function","metadata":{}},{"cell_type":"code","source":"@torch.no_grad()\ndef valid_one_epoch(model, dataloader, device, epoch):\n    model.eval()\n    \n    dataset_size = 0\n    running_loss= 0\n\n    bar = tqdm(enumerate(dataloader), total=len(dataloader))\n    for step, data in bar:\n        ids = data['input_ids'].to(device, dtype = torch.long)\n        mask = data['attention_mask'].to(device, dtype = torch.long)\n        targets = data['target'].to(device, dtype = torch.long)\n        \n        batch_size = ids.size(0)\n        outputs, output_type = model(ids, mask)\n        loss = criterion(outputs, targets)\n        \n        running_loss += (loss.item()*batch_size)\n        dataset_size += batch_size\n        \n        epoch_loss = running_loss / dataset_size\n#         bar.set_postfix(Epoch = epoch, Valid_loss = epoch_loss, LR=optimizer.param_groups[0]['lr'])\n        bar.set_postfix(Epoch = epoch, Valid_loss = epoch_loss)\n    gc.collect()\n    return epoch_loss\n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:18.014366Z","iopub.execute_input":"2022-08-02T18:54:18.014754Z","iopub.status.idle":"2022-08-02T18:54:18.031238Z","shell.execute_reply.started":"2022-08-02T18:54:18.014721Z","shell.execute_reply":"2022-08-02T18:54:18.029664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training process\n","metadata":{}},{"cell_type":"code","source":"def prepare_loaders(fold):\n    df_train = df[df['kfold'] != fold].reset_index(drop=True)\n    df_valid = df[df['kfold'] == fold].reset_index(drop=True)\n    \n    train_dataset = FeedbackDataset(df_train, tokenizer=CFG.tokenizer, max_length=CFG.max_length)\n    valid_dataset = FeedbackDataset(df_valid, tokenizer=CFG.tokenizer, max_length=CFG.max_length)\n\n    train_loader = DataLoader(train_dataset, batch_size=CFG.train_batch_size, collate_fn=collate_fn, \n                              num_workers=2, shuffle=True, pin_memory=True, drop_last=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=CFG.valid_batch_size, collate_fn=collate_fn,\n                              num_workers=2, shuffle=False, pin_memory=True)\n    \n    return train_loader, valid_loader","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:18.033344Z","iopub.execute_input":"2022-08-02T18:54:18.034085Z","iopub.status.idle":"2022-08-02T18:54:18.046420Z","shell.execute_reply.started":"2022-08-02T18:54:18.034044Z","shell.execute_reply":"2022-08-02T18:54:18.045400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_scheduler(optimizer):\n    if CFG.scheduler == 'CosineAnnealingLR':\n        scheduler = lr_scheduler.CosineAnnealingLR(optimizer, T_max=CFG.T_max, eta_min=CFG.min_lr)\n    elif CFG.scheduler == 'CosineAnnealingWarmRestarts':\n        scheduler = lr_scheduler.CosineAnnealingWarmRestarts(optimizer, T_0 = CFG.T_0, eta_min=CFG.min_lr)\n    elif CFG.scheduler == None:\n        return None\n    return scheduler","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:18.050789Z","iopub.execute_input":"2022-08-02T18:54:18.051529Z","iopub.status.idle":"2022-08-02T18:54:18.058955Z","shell.execute_reply.started":"2022-08-02T18:54:18.051501Z","shell.execute_reply":"2022-08-02T18:54:18.057977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @track_emissions\ndef run_training(model, device, num_epochs, fold, train_loader, valid_loader):\n    wandb.watch(model, log_freq = 100)\n    \n    start = time.time()\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_epoch_loss = np.inf\n    history = defaultdict(list)\n    for epoch in range(1,num_epochs+1):\n        gc.collect()\n        train_epoch_loss = train_one_epoch(model,train_loader, device, epoch)\n        valid_epoch_loss = valid_one_epoch(model, valid_loader, device, epoch)\n        \n        history['Train Loss'].append(train_epoch_loss)\n        history['Eval Loss'].append(valid_epoch_loss)\n        \n        wandb.log({'Train Loss': train_epoch_loss})\n        wandb.log({'Eval Loss': valid_epoch_loss})\n        \n        if valid_epoch_loss <= best_epoch_loss:\n            print(f\"Valid Loss Improved: {best_epoch_loss} -------> {valid_epoch_loss}\")\n            best_epoch_loss = valid_epoch_loss\n            run.summary['Best Loss']= best_epoch_loss\n            best_model_wts = copy.deepcopy(model.state_dict())\n            path = f'LossFold-{fold}.bin'\n            torch.save(model.state_dict(), path)\n            print('Model Saved')\n        \n    end = time.time()\n    time_eclipsed = end-start\n    print(f'Time complete in: {time_eclipsed//3600}h:{(time_eclipsed%3600)//60}m:{time_eclipsed%60}s')\n    print(f'Best Loss: {best_epoch_loss}')\n    \n    model.load_state_dict(best_model_wts)\n    \n    return model, history\n            ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:18.065959Z","iopub.execute_input":"2022-08-02T18:54:18.066804Z","iopub.status.idle":"2022-08-02T18:54:18.077842Z","shell.execute_reply.started":"2022-08-02T18:54:18.066777Z","shell.execute_reply":"2022-08-02T18:54:18.076911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Run training","metadata":{}},{"cell_type":"code","source":"transformers.logging.set_verbosity_error()\nfor fold in range(CFG.n_fold):\n    print(f'================ Fold: {fold} =================')\n    \n    cfg = dict(CFG.__dict__)\n    del cfg['__dict__'], cfg['__weakref__']\n    run = wandb.init(\n        project = 'FeedBack',\n        config = cfg,\n        job_type = 'Train',\n        group = CFG.group,\n        tags = [CFG.model_name, CFG.wandb_id],\n        name = f'{CFG.wandb_id}-Fold-{fold}',\n        anonymous='must'\n    )\n    \n    train_loader, valid_loader = prepare_loaders(fold)\n    model = FeedbackModel(CFG.model_name)\n    model.to(CFG.device)\n\n    if torch.cuda.is_available():\n        print(\"[INFO] Using GPU: {}\\n\".format(torch.cuda.get_device_name()))\n\n    model.set_optimizer_scheduler(\"Adam8bit\")\n            \n    model, history = run_training(model, CFG.device, CFG.epoch, fold, train_loader, valid_loader)\n    \n    run.finish()\n    gc.collect()          ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:54:18.178978Z","iopub.execute_input":"2022-08-02T18:54:18.179306Z","iopub.status.idle":"2022-08-02T18:55:28.286442Z","shell.execute_reply.started":"2022-08-02T18:54:18.179278Z","shell.execute_reply":"2022-08-02T18:55:28.285309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizations","metadata":{}},{"cell_type":"code","source":"url = f\"https://wandb.ai/phqlong/FeedBack/groups/{CFG.group}/\"\n\n# This is just to display the W&B run page in this interactive session.\nfrom IPython import display\n\n# we create an IFrame and set the width and height\niF = display.IFrame(url, width=1080, height=720)\niF","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:28.288791Z","iopub.execute_input":"2022-08-02T18:55:28.289535Z","iopub.status.idle":"2022-08-02T18:55:28.298078Z","shell.execute_reply.started":"2022-08-02T18:55:28.289490Z","shell.execute_reply":"2022-08-02T18:55:28.296891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Inference\n","metadata":{}},{"cell_type":"code","source":"import warnings,transformers,logging,torch\n\nwarnings.simplefilter('ignore')\nlogging.disable(logging.WARNING)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:28.300039Z","iopub.execute_input":"2022-08-02T18:55:28.300486Z","iopub.status.idle":"2022-08-02T18:55:28.307828Z","shell.execute_reply.started":"2022-08-02T18:55:28.300421Z","shell.execute_reply":"2022-08-02T18:55:28.306792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay_test(essay_id):\n    path = os.path.join(TEST_DIR, f'{essay_id}.txt')\n    essay_text = open(path, 'r').read()\n    return essay_text\n\ntest_df = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\n\ntest_df['essay_text']= test_df['essay_id'].apply(get_essay_test)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:28.310724Z","iopub.execute_input":"2022-08-02T18:55:28.311878Z","iopub.status.idle":"2022-08-02T18:55:28.342562Z","shell.execute_reply.started":"2022-08-02T18:55:28.311830Z","shell.execute_reply":"2022-08-02T18:55:28.341563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeedbackTestDataset(Dataset):\n    def __init__(self,df, max_length, tokenizer):\n        self.df = df\n        self.max_len = max_length\n        self.tokenizer = tokenizer\n        self.discourse_type = self.df['discourse_type'].values\n        self.discourse_text = self.df['discourse_text'].values\n        self.essays = self.df['essay_text'].values\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        discourse_type = self.discourse_type[index]\n        discourse_text = self.discourse_text[index]\n        essay = self.essays[index]\n\n        inputs = self.tokenizer.encode_plus(\n            f\"{discourse_type} {self.tokenizer.sep_token} {discourse_text}\", \n            essay,\n            truncation = True,\n            add_special_tokens = True, \n            max_length = self.max_len\n        )\n        \n        samples = {\n            'input_ids': inputs['input_ids'],\n            'attention_mask': inputs['attention_mask'],\n        }\n\n        if 'token_type_ids' in inputs:\n            samples['token_type_ids'] = inputs['token_type_ids']\n        \n        return samples\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:28.344568Z","iopub.execute_input":"2022-08-02T18:55:28.345177Z","iopub.status.idle":"2022-08-02T18:55:28.355158Z","shell.execute_reply.started":"2022-08-02T18:55:28.345134Z","shell.execute_reply":"2022-08-02T18:55:28.354164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_fn = DataCollatorWithPadding(tokenizer=CFG.tokenizer)\n\nsoftmax = nn.Softmax(dim=1)\nmodel = FeedbackModel(CFG.model_name)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:28.357011Z","iopub.execute_input":"2022-08-02T18:55:28.357685Z","iopub.status.idle":"2022-08-02T18:55:30.995511Z","shell.execute_reply.started":"2022-08-02T18:55:28.357644Z","shell.execute_reply":"2022-08-02T18:55:30.994431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_test_loader(test_df):    \n    test_dataset = FeedbackDataset(test_df, \n                                   tokenizer=CFG.tokenizer, \n                                   max_length=CFG.max_length,\n                                   training=False)\n    \n    test_loader = DataLoader(test_dataset, \n                             batch_size=CFG.valid_batch_size, \n                             collate_fn=collate_fn, \n                             num_workers=2, \n                             shuffle=False, \n                             pin_memory=True, \n                             drop_last=False)\n    return test_loader\n\ntest_loader = prepare_test_loader(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:30.997529Z","iopub.execute_input":"2022-08-02T18:55:30.997812Z","iopub.status.idle":"2022-08-02T18:55:31.005655Z","shell.execute_reply.started":"2022-08-02T18:55:30.997773Z","shell.execute_reply":"2022-08-02T18:55:31.004624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@torch.no_grad()\ndef inference(test_loader, model, device):\n    preds = []\n    model.eval()\n    model.to(device)\n    \n    bar = tqdm(enumerate(test_loader), total=len(test_loader))\n    \n    for step, data in bar: \n        ids = data['input_ids'].to(device, dtype = torch.long)\n        mask = data['attention_mask'].to(device, dtype = torch.long)\n        \n        output, _ = model(ids, mask)\n        y_preds = softmax(torch.tensor(output.to('cpu'))).numpy()\n        \n        preds.append(y_preds)\n         \n    predictions = np.concatenate(preds)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:31.007047Z","iopub.execute_input":"2022-08-02T18:55:31.008068Z","iopub.status.idle":"2022-08-02T18:55:31.018733Z","shell.execute_reply.started":"2022-08-02T18:55:31.008016Z","shell.execute_reply":"2022-08-02T18:55:31.017677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deberta_predictions = []\n\n\nfor fold in range(CFG.n_fold):\n    print(\"Fold {}\".format(fold))\n    \n    state = torch.load(f'/kaggle/working/LossFold-{fold}.bin')\n    model.load_state_dict(state)\n    \n    prediction = inference(test_loader, model, CFG.device)\n    deberta_predictions.append(prediction)\n    del state, prediction; gc.collect()\n    torch.cuda.empty_cache()\ndel model","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:31.020845Z","iopub.execute_input":"2022-08-02T18:55:31.021623Z","iopub.status.idle":"2022-08-02T18:55:35.702032Z","shell.execute_reply.started":"2022-08-02T18:55:31.021586Z","shell.execute_reply":"2022-08-02T18:55:35.700811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deberta_predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:35.708308Z","iopub.execute_input":"2022-08-02T18:55:35.708608Z","iopub.status.idle":"2022-08-02T18:55:35.727026Z","shell.execute_reply.started":"2022-08-02T18:55:35.708567Z","shell.execute_reply":"2022-08-02T18:55:35.720849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = np.mean(deberta_predictions, axis=0)\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:35.729883Z","iopub.execute_input":"2022-08-02T18:55:35.730233Z","iopub.status.idle":"2022-08-02T18:55:35.755918Z","shell.execute_reply.started":"2022-08-02T18:55:35.730194Z","shell.execute_reply":"2022-08-02T18:55:35.749703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = \"../input/feedback-prize-effectiveness/\"\nsubmission = pd.read_csv(os.path.join(INPUT_DIR, 'sample_submission.csv'))\n\nsubmission['Adequate'] = predictions[:, 0]\nsubmission['Effective'] = predictions[:, 1]\nsubmission['Ineffective'] = predictions[:, 2]\n\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:35.758342Z","iopub.execute_input":"2022-08-02T18:55:35.758655Z","iopub.status.idle":"2022-08-02T18:55:35.851430Z","shell.execute_reply.started":"2022-08-02T18:55:35.758618Z","shell.execute_reply":"2022-08-02T18:55:35.850137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T18:55:35.854092Z","iopub.execute_input":"2022-08-02T18:55:35.856415Z","iopub.status.idle":"2022-08-02T18:55:35.870342Z","shell.execute_reply.started":"2022-08-02T18:55:35.856368Z","shell.execute_reply":"2022-08-02T18:55:35.868430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}