{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom glob import glob\nimport gc\ngc.enable()\nimport matplotlib.pyplot as plt\nimport torch; print(\"\\n \\t...PyTorch Version: \", torch.__version__)\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms as T\nimport torchvision; print(\"\\n\\t...TorchVision Version: \", torchvision.__version__)\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom PIL import Image\nimport cv2\nimport albumentations as A\nimport time\nimport os\nimport copy\nfrom tqdm import tqdm\n\nfrom torch.cuda import amp\nscaler = amp.GradScaler()\nfrom torch.autograd import Variable\nimport torch.nn.functional as F\n\nimport numba\nimport numpy as np; print(\"\\n\\t...numpy version: \", np.__version__)\nfrom math import sqrt\nfrom scipy.spatial.distance import directed_hausdorff\nfrom scipy.ndimage import convolve\nfrom scipy.ndimage.morphology import distance_transform_edt as edt\n\nfrom torch.optim.lr_scheduler import _LRScheduler\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport unicodedata\nfrom glob import glob\nimport gc\nfrom sklearn.model_selection import KFold,GroupKFold,StratifiedKFold,StratifiedGroupKFold\ngc.enable()\n\nprint(\"\\n\\t...Import Finished\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:10.952551Z","iopub.execute_input":"2022-07-16T14:59:10.953226Z","iopub.status.idle":"2022-07-16T14:59:10.966784Z","shell.execute_reply.started":"2022-07-16T14:59:10.953189Z","shell.execute_reply":"2022-07-16T14:59:10.965799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy requests nlpaug","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:11.187187Z","iopub.execute_input":"2022-07-16T14:59:11.188101Z","iopub.status.idle":"2022-07-16T14:59:21.490493Z","shell.execute_reply.started":"2022-07-16T14:59:11.188065Z","shell.execute_reply":"2022-07-16T14:59:21.489286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install nltk>=3.4.5","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:21.493217Z","iopub.execute_input":"2022-07-16T14:59:21.493629Z","iopub.status.idle":"2022-07-16T14:59:31.027530Z","shell.execute_reply.started":"2022-07-16T14:59:21.493590Z","shell.execute_reply":"2022-07-16T14:59:31.026328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nlpaug.augmenter.word as naw\nimport nlpaug.augmenter.sentence as nas","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:31.029522Z","iopub.execute_input":"2022-07-16T14:59:31.030559Z","iopub.status.idle":"2022-07-16T14:59:31.036310Z","shell.execute_reply.started":"2022-07-16T14:59:31.030519Z","shell.execute_reply":"2022-07-16T14:59:31.035183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    model_nm            = '../input/deberta-v3-base/deberta-v3-base'\n    MAX_LEN             = 512\n    train_batch_size    = 16\n    valid_batch_size    = 32\n    num_labels          = 3\n    epochs              = 4\n    learning_rate       = 1e-5\n    min_lr              = 1e-6\n    T_max               = 500\n    weight_decay        = 1e-6\n    n_fold              = 5\n    pos_weight          = []\n    device              = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n    augmentation_p      = 0.05\n    augmentation_list   = [\"Lead\", \"Position\", \"Counterclaim\", \"Rebuttal\", \"Concluding Statement\"]#, \"Evidence\", \"Claim\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:31.037935Z","iopub.execute_input":"2022-07-16T14:59:31.038645Z","iopub.status.idle":"2022-07-16T14:59:31.046815Z","shell.execute_reply.started":"2022-07-16T14:59:31.038610Z","shell.execute_reply":"2022-07-16T14:59:31.045530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/feedback-prize-effectivenessprocesseddata/train_processed.csv')\ntrain_df['text_content'] = train_df['text_content'].apply(lambda x : unicodedata.normalize('NFKD', x).strip())\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:31.051642Z","iopub.execute_input":"2022-07-16T14:59:31.053032Z","iopub.status.idle":"2022-07-16T14:59:33.538120Z","shell.execute_reply.started":"2022-07-16T14:59:31.052967Z","shell.execute_reply":"2022-07-16T14:59:33.537227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_effective = train_df['discourse_effectiveness'].value_counts()\nineffective_count = d_effective.loc[0]\nadequate_count = d_effective.loc[1]\neffective_count = d_effective.loc[2]\neffectiveness_samples = [ineffective_count, adequate_count, effective_count]\nprint(\"[Ineffective, Adequate, Effective] : \" , effectiveness_samples, '\\n')\nnormedWeights = [ 1 - (x / sum(effectiveness_samples)) for x in effectiveness_samples]\nCFG.pos_weight = normedWeights\nprint(\"Weights normalized for unbalanced classes : \", CFG.pos_weight)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:33.539628Z","iopub.execute_input":"2022-07-16T14:59:33.539955Z","iopub.status.idle":"2022-07-16T14:59:33.549534Z","shell.execute_reply.started":"2022-07-16T14:59:33.539922Z","shell.execute_reply":"2022-07-16T14:59:33.548301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BuildDataSet(torch.utils.data.Dataset):\n    def __init__(self, df, tokenizer):\n        self.df = df\n        self.tokenizer = tokenizer\n        self.discourse_type = df[\"discourse_type\"].values\n        self.discourse_text = df[\"discourse_text\"].values\n        self.text_content = df[\"text_content\"].values\n        self.labels = df[\"discourse_effectiveness\"].values\n    \n    def __len__(self):\n        return len(self.df)\n    \n    \n    def __getitem__(self, index):\n        discourse_type = self.discourse_type[index]\n        discourse_text = self.discourse_text[index]\n        text_content = self.text_content[index]\n        label = self.labels[index]\n        \n        \n        text = discourse_type + self.tokenizer.sep_token + discourse_text + self.tokenizer.sep_token + text_content\n        input_ = self.tokenizer.encode_plus(text,\n                                            None,\n                                            truncation = True,\n                                            add_special_tokens = True,\n                                            max_length= CFG.MAX_LEN,\n                                            pad_to_max_length=True)\n        # change dtype to float\n        ids            = torch.tensor(input_['input_ids'], dtype=torch.long)\n        mask           = torch.tensor(input_['attention_mask'], dtype=torch.long)\n        token_type_ids = torch.tensor(input_['token_type_ids'], dtype=torch.long)\n        \n        label          = torch.tensor(label)\n        \n        \"\"\"\n        label as one hot encoding for BCEWithLogitLoss()\n        \"\"\"\n        label_OH = F.one_hot(label, num_classes = CFG.num_labels).to(torch.float)\n        \n        \n        return {\n            \"input_ids\" : ids,\n            \"token_type_ids\" : token_type_ids,\n            \"attention_masks\" : mask,\n            \"labels\" : label_OH\n        }\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:33.551883Z","iopub.execute_input":"2022-07-16T14:59:33.552910Z","iopub.status.idle":"2022-07-16T14:59:33.565175Z","shell.execute_reply.started":"2022-07-16T14:59:33.552866Z","shell.execute_reply":"2022-07-16T14:59:33.564266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, dataloaders, optimizer, scheduler, k_fold, scaler, num_epoch = CFG.epochs):\n    \n    model.to(CFG.device)\n    since = time.time()\n    \n    best_model_wts = copy.deepcopy(model.state_dict())\n    #best_acc = 0.0\n    best_loss = 1.0\n    \n    #loss_fc  = torch.nn.BCEWithLogitsLoss(pos_weight = torch.tensor(CFG.pos_weight).to(torch.float).to(CFG.device))\n    \n    version = 0\n    \n    for epoch in range(1, num_epoch + 1):\n        print(f'Epoch {epoch}/{num_epoch}')\n        print('-' * 15)\n        \n        for phase in ['Train', 'Eval']:\n            if phase == 'Train':\n                model.train()\n            else:\n                model.eval()\n            \n            print(f'{phase} : ')    \n            running_loss = 0.0\n            running_corrects = 0\n            tqdm_bar = tqdm(enumerate(dataloaders[phase]), total = (len(dataloaders[phase])))\n            sample_nums = 0\n            for batch_num, inputs in tqdm_bar:\n                \n                input_ids      = inputs['input_ids'].to(CFG.device)\n                attention_mask = inputs['attention_masks'].to(CFG.device)\n                token_type_ids = inputs['token_type_ids'].to(CFG.device)\n                labels         = inputs['labels'].to(CFG.device)\n    \n                with torch.set_grad_enabled(phase == 'Train'):            \n                    if phase == 'Train':\n                        with torch.cuda.amp.autocast():\n                            output = model(input_ids, attention_mask, token_type_ids, labels = labels)\n                            logits = output.logits\n                            loss   = output.loss\n                            #loss   = loss_fc(logits, labels)\n                        optimizer.zero_grad()\n                        scaler.scale(loss).backward()\n                        scaler.step(optimizer)\n                        scaler.update()\n                    else:\n                        output = model(input_ids, attention_mask, token_type_ids, labels = labels)\n                        loss   = output.loss\n            \n                running_loss += loss.item() * input_ids.size(0)\n                \"\"\"\n                Compute Metrics\n                \"\"\"\n                # output is the raw output, apply a sigmoid to retrieve probabilities\n                logits   = output.logits\n                probs    = torch.nn.Sigmoid()(logits)\n                preds    = torch.argmax(probs, dim = 1)\n                labels_  = torch.argmax(labels, dim = 1) # be 0/1/2 because [0,1,0] or [1,0,0] or [0,0,1]\n                corrects = torch.sum(preds == labels_.data)\n                running_corrects += corrects\n                \n                sample_nums += input_ids.size(0)\n                accumulated_loss = running_loss / sample_nums\n                accumulated_acc  = running_corrects / sample_nums\n                \n                tqdm_bar.set_description(desc=f\"loss: {accumulated_loss:.4f}, acc : {accumulated_acc:.4f}, Learning rate : {optimizer.param_groups[0]['lr']}\")\n            \n            \n            if phase == 'Train':\n                scheduler.step()\n        \n            \"\"\"\n            epoch Metric \n            \"\"\"\n            epoch_loss = running_loss / dataset_size[phase]\n            epoch_acc  = running_corrects.double() / dataset_size[phase]\n            print(f'{phase} Loss : {epoch_loss: .4f} Acc: {epoch_acc: .4f}')\n        \n            \"\"\"\n            Deep copy and save a better model\n            \"\"\"\n            #if phase == 'Eval' and epoch_acc > best_acc :\n            if phase == 'Eval' and epoch_loss < best_loss :\n                best_loss = epoch_loss\n                best_model_wts = copy.deepcopy(model.state_dict())\n                print(f\"Saving a better model in Fold : {k_fold}  with lower loss: {best_loss:.4f}\")\n                version = version + 1\n                torch.save(best_model_wts, f'./best_weight_DeBERTa_v3_base_fold{k_fold}_ver{version}.pt')\n            del input_ids, attention_mask, token_type_ids, labels\n           \n        gc.collect()\n        torch.cuda.empty_cache()\n        print()\n    time_elasped = time.time() - since\n    \n    model.load_state_dict(best_model_wts)\n    \n    return model  ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:33.567223Z","iopub.execute_input":"2022-07-16T14:59:33.567702Z","iopub.status.idle":"2022-07-16T14:59:33.589548Z","shell.execute_reply.started":"2022-07-16T14:59:33.567664Z","shell.execute_reply":"2022-07-16T14:59:33.588425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:33.590757Z","iopub.execute_input":"2022-07-16T14:59:33.591152Z","iopub.status.idle":"2022-07-16T14:59:34.029529Z","shell.execute_reply.started":"2022-07-16T14:59:33.591116Z","shell.execute_reply":"2022-07-16T14:59:34.028382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModelForSequenceClassification, AutoTokenizer, DataCollatorWithPadding\n\ntokenizer = AutoTokenizer.from_pretrained(CFG.model_nm)\n\n#collate_fn = DataCollatorWithPadding(tokenizer=tokenizer)\n#lr_scheduler = torch.optim.lr_scheduler.CosineAnnealingWarmRestarts(optimizer, T_0 = 20, eta_min = 1e-5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:34.030757Z","iopub.execute_input":"2022-07-16T14:59:34.031050Z","iopub.status.idle":"2022-07-16T14:59:34.760662Z","shell.execute_reply.started":"2022-07-16T14:59:34.031025Z","shell.execute_reply":"2022-07-16T14:59:34.759536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#AutoModelForSequenceClassification.from_pretrained(CFG.model_nm, num_labels=3, problem_type = \"multi_label_classification\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:34.762015Z","iopub.execute_input":"2022-07-16T14:59:34.762485Z","iopub.status.idle":"2022-07-16T14:59:34.767611Z","shell.execute_reply.started":"2022-07-16T14:59:34.762447Z","shell.execute_reply":"2022-07-16T14:59:34.766317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:34.781202Z","iopub.execute_input":"2022-07-16T14:59:34.781745Z","iopub.status.idle":"2022-07-16T14:59:34.786999Z","shell.execute_reply.started":"2022-07-16T14:59:34.781710Z","shell.execute_reply":"2022-07-16T14:59:34.785769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug_percent = 0.8\n\nsyn_aug = naw.ContextualWordEmbsAug(model_path='roberta-base', action=\"substitute\", aug_p = 0.3, device = 'cuda')\n\ninsert_aug = naw.ContextualWordEmbsAug(model_path='roberta-base', action=\"insert\", aug_p = 0.3, device = 'cuda')\n\nswap_aug = naw.RandomWordAug(action=\"swap\", aug_p = 0.3)\n\ndelete_aug =naw.RandomWordAug(action=\"delete\", aug_p = 0.3)\n\n#abstract_sum = nas.AbstSummAug(model_path='facebook/bart-large-cnn' , device=\"cuda\")\n\naug_tools = [syn_aug, insert_aug, swap_aug, delete_aug]\n\n\ndef text_augmentation(text):\n    res = []\n    pick = random.choice(range(0, 4))\n    #if pick < 4:\n    aug = aug_tools[pick]\n    return aug.augment(text)\n    #else:\n        #return [text]\n    #return syn_aug.augment(text) + insert_aug.augment(text) + swap_aug.augment(text) + delete_aug.augment(text)\n    #return swap_aug.augment(text) + delete_aug.augment(text)\n\n    \ndef augmentation(df):\n    section = int(len(df) * aug_percent)\n    getAugmentation(df[:section])\n    return df\n\ndef getAugmentation(df):\n    print(\"Augmentation starts...\\n\")\n    temp_df = pd.DataFrame()\n    tqdm_bar = tqdm(df.iterrows(), total = (len(df)))\n    for index, row in tqdm_bar:\n       \n        if row.discourse_type in CFG.augmentation_list:\n            augmented_discourse_text    = text_augmentation(row.discourse_text)\n            #print(augmented_discourse_text)\n            augmented_discourse_content = text_augmentation(row.text_content)\n            \"\"\"for i in range(len(augmented_discourse_text)):\n                augmented_row = {\"discourse_id\" : row.discourse_id, \n                                \"essay_id\" : row.essay_id,\n                                \"discourse_text\": augmented_discourse_text[i],\n                                \"discourse_type\": row.discourse_type,\n                                 'discourse_effectiveness': row.discourse_effectiveness,\n                                \"text_content\": row.text_content }#augmented_discourse_content[i]}\n                temp_df = temp_df.append(augmented_row, ignore_index = True)\"\"\"\n            for i in range(len(augmented_discourse_text)):\n                df.at[index,'text_content'] = augmented_discourse_content[i]\n                df.at[index,'discourse_text'] = augmented_discourse_text[i]\n                \n    \"\"\"temp_df = temp_df.astype({'discourse_effectiveness': 'int64'})\n    return pd.concat([df, temp_df], ignore_index = True)\"\"\"\n    \n    return df\n\n                \n\"\"\"discourse_type = train_df.iloc[0].discourse_type\ndiscourse_text = train_df.iloc[0].discourse_text\ntext_content   = train_df.iloc[0].text_content\n\"\"\"\n#text = discourse_type + tokenizer.sep_token + discourse_text + tokenizer.sep_token + text_content\n#text = train_df.iloc[0].discourse_text\ntext_df = train_df[:11].copy()\naug_df = augmentation(text_df)\n#aug_df['text_content'] = aug_df['text_content'].apply(lambda x : unicodedata.normalize('NFKD', x).strip())\n#aug_df['discourse_text'] = aug_df['discourse_text'].apply(lambda x : unicodedata.normalize('NFKD', x).strip())\naug_df","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:34.792764Z","iopub.execute_input":"2022-07-16T14:59:34.793123Z","iopub.status.idle":"2022-07-16T14:59:35.213075Z","shell.execute_reply.started":"2022-07-16T14:59:34.793089Z","shell.execute_reply":"2022-07-16T14:59:35.212034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createKFold(df):\n    SKF = StratifiedKFold(n_splits = CFG.n_fold)\n    df[\"kfold\"] = -1\n    for f, (train_index, test_index) in enumerate(SKF.split(df['discourse_id'], df['discourse_effectiveness'])):\n        df.loc[test_index, 'kfold'] = f\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:35.230747Z","iopub.execute_input":"2022-07-16T14:59:35.231107Z","iopub.status.idle":"2022-07-16T14:59:35.251076Z","shell.execute_reply.started":"2022-07-16T14:59:35.231072Z","shell.execute_reply":"2022-07-16T14:59:35.250059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:35.252428Z","iopub.execute_input":"2022-07-16T14:59:35.252869Z","iopub.status.idle":"2022-07-16T14:59:35.261901Z","shell.execute_reply.started":"2022-07-16T14:59:35.252832Z","shell.execute_reply":"2022-07-16T14:59:35.260950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SKF = StratifiedKFold(n_splits = CFG.n_fold)\nk_fold = 1\nfor train_index, test_index in SKF.split(train_df['discourse_id'], train_df['discourse_effectiveness']): #, train_df['discourse_type']):\n    \n    # Data preparation\n    #sub_train_df = train_df.iloc[train_index]\n    sub_train_df = augmentation(train_df.iloc[train_index].copy())\n    sub_valid_df = train_df.iloc[test_index]\n    \n    train_dataset    = BuildDataSet(sub_train_df, tokenizer)\n    train_dataloader = DataLoader(train_dataset, batch_size = CFG.train_batch_size, num_workers = 2, shuffle = True, pin_memory = True, drop_last = True)\n    valid_dataset    = BuildDataSet(sub_valid_df, tokenizer)\n    valid_dataloader = DataLoader(valid_dataset, batch_size = CFG.valid_batch_size, num_workers = 2, shuffle = True, pin_memory = True)\n    \n    dataset_size = {\"Train\": len(train_dataset), \"Eval\": len(valid_dataset)}\n    print(f\"Fold : {k_fold}, training set : {len(train_dataset)}, validation set : {len(valid_dataset)} \\n\")\n    dataloaders = {\"Train\" : train_dataloader, \"Eval\" : valid_dataloader}\n    \n    # Model Configuration\n    model = AutoModelForSequenceClassification.from_pretrained(CFG.model_nm, num_labels=3, problem_type = \"multi_label_classification\")\n    \n    optimizer = torch.optim.AdamW(model.parameters(), lr = CFG.learning_rate, weight_decay = CFG.weight_decay)\n\n    lr_scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer,T_max= CFG.T_max, \n                                                   eta_min=CFG.min_lr)\n    \n    # Start Training\n    train_model(model, dataloaders, optimizer, lr_scheduler, k_fold, scaler = torch.cuda.amp.GradScaler(), num_epoch = CFG.epochs)\n    \n    k_fold = k_fold + 1\n    if k_fold == 3:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-07-16T14:59:35.263194Z","iopub.execute_input":"2022-07-16T14:59:35.264139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}