{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport copy\nimport time\nimport random\nimport string\nimport joblib\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\n# Utils\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import GroupKFold, KFold\n\n# For Transformer Models\nfrom transformers import AutoTokenizer, AutoModel, AutoConfig, AdamW\nfrom transformers import DataCollatorWithPadding\n# from transformers import DebertaModel, DebertaConfig, DebertaTokenizer, DebertaV2Tokenizer, DebertaV2Model, DebertaV2Config\n\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\n# os.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T21:54:46.436377Z","iopub.execute_input":"2022-07-18T21:54:46.436866Z","iopub.status.idle":"2022-07-18T21:54:54.361513Z","shell.execute_reply.started":"2022-07-18T21:54:46.436781Z","shell.execute_reply":"2022-07-18T21:54:54.360534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = \"../input/feedback-prize-effectiveness/train\"\nTEST_DIR = \"../input/feedback-prize-effectiveness/test\"","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:54:54.364048Z","iopub.execute_input":"2022-07-18T21:54:54.364733Z","iopub.status.idle":"2022-07-18T21:54:54.370015Z","shell.execute_reply.started":"2022-07-18T21:54:54.364692Z","shell.execute_reply":"2022-07-18T21:54:54.369031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n\nCONFIG = {\"seed\": 1997,\n          \"epochs\": 4,\n          \"model_name\": \"lsanochkin/deberta-large-feedback\",\n#           \"model_name\": \"microsoft/deberta-v3-large\",\n          \"train_batch_size\": 2,\n          \"valid_batch_size\": 16,\n          \"max_length\": 512,\n          \"learning_rate\": 1e-5,\n          \"scheduler\": 'CosineAnnealingLR',\n          \"min_lr\": 1e-6,\n          \"T_max\": 500,\n          \"weight_decay\": 1e-6,\n          \"n_accumulate\": 1,\n          \"num_classes\": 3,\n#           \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n          \"competition\": \"FeedBack\",\n#           \"_wandb_kernel\": \"deb\",\n          }\n\nCONFIG[\"tokenizer\"] = AutoTokenizer.from_pretrained(CONFIG['model_name'])\nmodel = AutoModel.from_pretrained(CONFIG['model_name'])\nconfig = AutoConfig.from_pretrained(CONFIG['model_name'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:54:54.371577Z","iopub.execute_input":"2022-07-18T21:54:54.372382Z","iopub.status.idle":"2022-07-18T21:56:21.959049Z","shell.execute_reply.started":"2022-07-18T21:54:54.372342Z","shell.execute_reply":"2022-07-18T21:56:21.958143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG['seed'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:21.960513Z","iopub.execute_input":"2022-07-18T21:56:21.960845Z","iopub.status.idle":"2022-07-18T21:56:21.969943Z","shell.execute_reply.started":"2022-07-18T21:56:21.960810Z","shell.execute_reply":"2022-07-18T21:56:21.969108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay(essay_id):\n    essay_path = os.path.join(TRAIN_DIR, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n\ndf = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ndf['essay_text'] = df['essay_id'].apply(get_essay)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:21.973881Z","iopub.execute_input":"2022-07-18T21:56:21.974303Z","iopub.status.idle":"2022-07-18T21:56:53.264589Z","shell.execute_reply.started":"2022-07-18T21:56:21.974273Z","shell.execute_reply":"2022-07-18T21:56:53.263653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder()\ndf['discourse_effectiveness'] = encoder.fit_transform(df['discourse_effectiveness'])\n\nwith open(\"le.pkl\", \"wb\") as fp:\n    joblib.dump(encoder, fp)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.266013Z","iopub.execute_input":"2022-07-18T21:56:53.266415Z","iopub.status.idle":"2022-07-18T21:56:53.286615Z","shell.execute_reply.started":"2022-07-18T21:56:53.266369Z","shell.execute_reply":"2022-07-18T21:56:53.285575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeedBackDataset(Dataset):\n    def __init__(self, df, tokenizer, max_length):\n        self.df = df\n        self.max_len = max_length\n        self.tokenizer = tokenizer\n        self.discourse = df['discourse_text'].values\n        self.essay = df['essay_text'].values\n        self.targets = df['discourse_effectiveness'].values\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        discourse = self.discourse[index]\n        essay = self.essay[index]\n        text = discourse + \" \" + self.tokenizer.sep_token + \" \" + essay\n        inputs = self.tokenizer.encode_plus(\n                        text,\n                        truncation=True,\n                        add_special_tokens=True,\n                        max_length=self.max_len\n                    )\n        \n        return {\n            'input_ids': inputs['input_ids'],\n            'attention_mask': inputs['attention_mask'],\n            'target': self.targets[index]\n        }","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.288016Z","iopub.execute_input":"2022-07-18T21:56:53.288447Z","iopub.status.idle":"2022-07-18T21:56:53.299462Z","shell.execute_reply.started":"2022-07-18T21:56:53.288410Z","shell.execute_reply":"2022-07-18T21:56:53.298498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_fn = DataCollatorWithPadding(tokenizer=CONFIG['tokenizer'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.300590Z","iopub.execute_input":"2022-07-18T21:56:53.300846Z","iopub.status.idle":"2022-07-18T21:56:53.308370Z","shell.execute_reply.started":"2022-07-18T21:56:53.300815Z","shell.execute_reply":"2022-07-18T21:56:53.307434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MeanPooling(nn.Module):\n    def __init__(self):\n        super(MeanPooling, self).__init__()\n        \n    def forward(self, last_hidden_state, attention_mask):\n        input_mask_expanded = attention_mask.unsqueeze(-1).expand(last_hidden_state.size()).float()\n        sum_embeddings = torch.sum(last_hidden_state * input_mask_expanded, 1)\n        sum_mask = input_mask_expanded.sum(1)\n        sum_mask = torch.clamp(sum_mask, min=1e-9)\n        mean_embeddings = sum_embeddings / sum_mask\n        return mean_embeddings\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.310011Z","iopub.execute_input":"2022-07-18T21:56:53.310346Z","iopub.status.idle":"2022-07-18T21:56:53.322582Z","shell.execute_reply.started":"2022-07-18T21:56:53.310313Z","shell.execute_reply":"2022-07-18T21:56:53.321662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass FeedBackModel(nn.Module):\n    def __init__(self):\n        super(FeedBackModel, self).__init__()\n        self.model = AutoModel.from_pretrained(CONFIG['model_name'])\n        self.config = AutoConfig.from_pretrained(CONFIG['model_name'])\n        self.drop = nn.Dropout(p=0.1)\n        self.pooler = MeanPooling()\n        self.fc = nn.Linear(self.config.hidden_size, CONFIG['num_classes'])\n        \n    def forward(self, ids, mask):        \n        out = self.model(input_ids=ids,attention_mask=mask,\n                         output_hidden_states=False)\n        out = self.pooler(out.last_hidden_state, mask)\n        out = self.drop(out)\n        outputs = self.fc(out)\n        return outputs","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.323807Z","iopub.execute_input":"2022-07-18T21:56:53.324302Z","iopub.status.idle":"2022-07-18T21:56:53.334367Z","shell.execute_reply.started":"2022-07-18T21:56:53.324264Z","shell.execute_reply":"2022-07-18T21:56:53.333446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = AutoModel.from_pretrained(CONFIG['model_name'])\n# config = AutoConfig.from_pretrained(CONFIG['model_name'])\nconfig.hidden_size","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.335717Z","iopub.execute_input":"2022-07-18T21:56:53.336218Z","iopub.status.idle":"2022-07-18T21:56:53.347306Z","shell.execute_reply.started":"2022-07-18T21:56:53.336182Z","shell.execute_reply":"2022-07-18T21:56:53.346161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# outputs = model(input_ids=ids,attention_mask=mask,\n#                      output_hidden_states=False)\n# outputs.last_hidden_state.size()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.349391Z","iopub.execute_input":"2022-07-18T21:56:53.349733Z","iopub.status.idle":"2022-07-18T21:56:53.354455Z","shell.execute_reply.started":"2022-07-18T21:56:53.349698Z","shell.execute_reply":"2022-07-18T21:56:53.353452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def criterion(outputs, labels):\n    return nn.CrossEntropyLoss()(outputs, labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.356205Z","iopub.execute_input":"2022-07-18T21:56:53.356694Z","iopub.status.idle":"2022-07-18T21:56:53.364007Z","shell.execute_reply.started":"2022-07-18T21:56:53.356659Z","shell.execute_reply":"2022-07-18T21:56:53.363014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_epoch(model, optimizer, scheduler, dataloader, device, epoch):\n    model.train()\n    \n    dataset_size = 0\n    running_loss = 0.0\n    \n    bar = tqdm(enumerate(dataloader), total=len(dataloader))\n    for step, data in bar:\n        ids = data['input_ids'].to(device, dtype = torch.long)\n        mask = data['attention_mask'].to(device, dtype = torch.long)\n        targets = data['target'].to(device, dtype=torch.long)\n        \n        batch_size = ids.size(0)\n\n        outputs = model(ids, mask)\n        \n        loss = criterion(outputs, targets)\n        loss = loss / CONFIG['n_accumulate']\n        loss.backward()\n    \n        if (step + 1) % CONFIG['n_accumulate'] == 0:\n            optimizer.step()\n\n            # zero the parameter gradients\n            optimizer.zero_grad()\n\n            if scheduler is not None:\n                scheduler.step()\n                \n        running_loss += (loss.item() * batch_size)\n        dataset_size += batch_size\n        \n        epoch_loss = running_loss / dataset_size\n        \n        bar.set_postfix(Epoch=epoch, Train_Loss=epoch_loss,\n                        LR=optimizer.param_groups[0]['lr'])\n    gc.collect()\n    \n    return epoch_loss","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.369338Z","iopub.execute_input":"2022-07-18T21:56:53.371087Z","iopub.status.idle":"2022-07-18T21:56:53.381593Z","shell.execute_reply.started":"2022-07-18T21:56:53.371061Z","shell.execute_reply":"2022-07-18T21:56:53.380583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef valid_one_epoch(model, dataloader, device, epoch):\n    \n    model.eval()\n    with torch.no_grad():\n\n        dataset_size = 0\n        running_loss = 0.0\n    \n        bar = tqdm(enumerate(dataloader), total=len(dataloader))\n        for step, data in bar:        \n            ids = data['input_ids'].to(device, dtype = torch.long)\n            mask = data['attention_mask'].to(device, dtype = torch.long)\n            targets = data['target'].to(device, dtype=torch.long)\n\n            batch_size = ids.size(0)\n\n            outputs = model(ids, mask)\n\n            loss = criterion(outputs, targets)\n\n            running_loss += (loss.item() * batch_size)\n            dataset_size += batch_size\n\n            epoch_loss = running_loss / dataset_size\n\n            bar.set_postfix(Epoch=epoch, Valid_Loss=epoch_loss,\n                            LR=optimizer.param_groups[0]['lr'])   \n\n        gc.collect()\n    \n    return epoch_loss","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.384731Z","iopub.execute_input":"2022-07-18T21:56:53.384997Z","iopub.status.idle":"2022-07-18T21:56:53.396281Z","shell.execute_reply.started":"2022-07-18T21:56:53.384954Z","shell.execute_reply":"2022-07-18T21:56:53.395413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_training(model, optimizer, scheduler, device, num_epochs):\n    # To automatically log gradients\n#     wandb.watch(model, log_freq=100)\n    \n    if torch.cuda.is_available():\n        print(\"[INFO] Using GPU: {}\\n\".format(torch.cuda.get_device_name()))\n    \n    start = time.time()\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_epoch_loss = np.inf\n    history = defaultdict(list)\n    \n    for epoch in range(1, num_epochs + 1): \n        gc.collect()\n        train_epoch_loss = train_one_epoch(model, optimizer, scheduler, \n                                           dataloader=train_loader, \n                                           device=device, epoch=epoch)\n        \n        val_epoch_loss = valid_one_epoch(model, valid_loader, device=device, \n                                         epoch=epoch)\n    \n        history['Train Loss'].append(train_epoch_loss)\n        history['Valid Loss'].append(val_epoch_loss)\n        \n        # Log the metrics\n#         wandb.log({\"Train Loss\": train_epoch_loss})\n#         wandb.log({\"Valid Loss\": val_epoch_loss})\n        \n        # deep copy the model\n        if val_epoch_loss <= best_epoch_loss:\n            print(f\"Validation Loss Improved ({best_epoch_loss} ---> {val_epoch_loss})\")\n            best_epoch_loss = val_epoch_loss\n#             run.summary[\"Best Loss\"] = best_epoch_loss\n            best_model_wts = copy.deepcopy(model.state_dict())\n            PATH = f\"Loss-Model.bin\"\n            torch.save(model.state_dict(), PATH)\n            # Save a model file from the current directory\n            print(f\"Model Saved\")\n            \n        print()\n    \n    end = time.time()\n    time_elapsed = end - start\n    print('Training complete in {:.0f}h {:.0f}m {:.0f}s'.format(\n        time_elapsed // 3600, (time_elapsed % 3600) // 60, (time_elapsed % 3600) % 60))\n    print(\"Best Loss: {:.4f}\".format(best_epoch_loss))\n    \n    # load best model weights\n    model.load_state_dict(best_model_wts)\n    \n    return model, history","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.398011Z","iopub.execute_input":"2022-07-18T21:56:53.398407Z","iopub.status.idle":"2022-07-18T21:56:53.411137Z","shell.execute_reply.started":"2022-07-18T21:56:53.398367Z","shell.execute_reply":"2022-07-18T21:56:53.410169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df.sample(frac=0.8,random_state=200)\ndf_valid = df.drop(df_train.index).reset_index(drop=True)\ndf_train = df_train.reset_index(drop=True)\n\n\ntrain_dataset = FeedBackDataset(df_train, tokenizer=CONFIG['tokenizer'], max_length=CONFIG['max_length'])\nvalid_dataset = FeedBackDataset(df_valid, tokenizer=CONFIG['tokenizer'], max_length=CONFIG['max_length'])\n\n\ntrain_loader = DataLoader(train_dataset, batch_size=CONFIG['train_batch_size'], collate_fn=collate_fn, \n                          num_workers=4, shuffle=True, pin_memory=True, drop_last=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=CONFIG['valid_batch_size'], collate_fn=collate_fn,\n                              num_workers=4, shuffle=False, pin_memory=True)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.412565Z","iopub.execute_input":"2022-07-18T21:56:53.413203Z","iopub.status.idle":"2022-07-18T21:56:53.450548Z","shell.execute_reply.started":"2022-07-18T21:56:53.413111Z","shell.execute_reply":"2022-07-18T21:56:53.449724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_scheduler(optimizer):\n    if CONFIG['scheduler'] == 'CosineAnnealingLR':\n        scheduler = lr_scheduler.CosineAnnealingLR(optimizer,T_max=CONFIG['T_max'], \n                                                   eta_min=CONFIG['min_lr'])\n    elif CONFIG['scheduler'] == 'CosineAnnealingWarmRestarts':\n        scheduler = lr_scheduler.CosineAnnealingWarmRestarts(optimizer,T_0=CONFIG['T_0'], \n                                                             eta_min=CONFIG['min_lr'])\n    elif CONFIG['scheduler'] == None:\n        return None\n        \n    return scheduler","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.451785Z","iopub.execute_input":"2022-07-18T21:56:53.452700Z","iopub.status.idle":"2022-07-18T21:56:53.459874Z","shell.execute_reply.started":"2022-07-18T21:56:53.452663Z","shell.execute_reply":"2022-07-18T21:56:53.458688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install objsize","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:56:53.463095Z","iopub.execute_input":"2022-07-18T21:56:53.463646Z","iopub.status.idle":"2022-07-18T21:57:04.160894Z","shell.execute_reply.started":"2022-07-18T21:56:53.463617Z","shell.execute_reply":"2022-07-18T21:57:04.159776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = FeedBackModel()\n\n#     model = model_base\nmodel.to(device)\n\n# Define Optimizer and Scheduler\noptimizer = AdamW(model.parameters(), lr=CONFIG['learning_rate'], weight_decay=CONFIG['weight_decay'])\nscheduler = fetch_scheduler(optimizer)\n\nmodel, history = run_training(model, optimizer, scheduler,\n                              device=device,\n                              num_epochs=1)\n\n    \n\ndel model, history#, train_loader, valid_loader\n_ = gc.collect()\nprint()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:57:04.162923Z","iopub.execute_input":"2022-07-18T21:57:04.163317Z","iopub.status.idle":"2022-07-18T21:57:44.285695Z","shell.execute_reply.started":"2022-07-18T21:57:04.163279Z","shell.execute_reply":"2022-07-18T21:57:44.283014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import objsize\n# objsize.get_deep_size(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:57:44.288784Z","iopub.status.idle":"2022-07-18T21:57:44.291959Z","shell.execute_reply.started":"2022-07-18T21:57:44.291696Z","shell.execute_reply":"2022-07-18T21:57:44.291723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install GPUtil\n\n# from GPUtil import showUtilization as gpu_usage\n# gpu_usage()   ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:57:44.295951Z","iopub.status.idle":"2022-07-18T21:57:44.298313Z","shell.execute_reply.started":"2022-07-18T21:57:44.298056Z","shell.execute_reply":"2022-07-18T21:57:44.298082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import torch\n# from GPUtil import showUtilization as gpu_usage\n# from numba import cuda\n\n# def free_gpu_cache():\n#     print(\"Initial GPU Usage\")\n#     gpu_usage()                             \n\n#     torch.cuda.empty_cache()\n\n#     cuda.select_device(0)\n#     cuda.close()\n#     cuda.select_device(0)\n\n#     print(\"GPU Usage after emptying the cache\")\n#     gpu_usage()\n\n# free_gpu_cache()                           ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:57:44.301333Z","iopub.status.idle":"2022-07-18T21:57:44.302241Z","shell.execute_reply.started":"2022-07-18T21:57:44.301931Z","shell.execute_reply":"2022-07-18T21:57:44.301960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}