{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 📝 Versions\n\nVersion 2: DeBertaV3 for Fold 4 **CV:0.  LB: 0.**\n","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Imports","metadata":{}},{"cell_type":"code","source":"# ! pip install cloud-tpu-client==0.10 https://storage.googleapis.com/tpu-pytorch/wheels/torch_xla-1.8-cp37-cp37m-linux_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:20.951667Z","iopub.execute_input":"2022-07-27T20:15:20.952629Z","iopub.status.idle":"2022-07-27T20:15:20.975501Z","shell.execute_reply.started":"2022-07-27T20:15:20.952512Z","shell.execute_reply":"2022-07-27T20:15:20.974659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom tqdm.auto import tqdm\n\nimport numpy as np \nimport pandas as pd \n\nfrom text_unidecode import unidecode\nfrom typing import Dict, List, Tuple\nimport codecs\n\nfrom sklearn.metrics import log_loss\n\nfrom transformers import AutoModel, AutoTokenizer, AdamW, DataCollatorWithPadding\n\nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\n\nimport pytorch_lightning as pl\nfrom pytorch_lightning import Trainer, seed_everything\nfrom pytorch_lightning.callbacks import ModelCheckpoint, EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:20.991948Z","iopub.execute_input":"2022-07-27T20:15:20.992493Z","iopub.status.idle":"2022-07-27T20:15:30.740938Z","shell.execute_reply.started":"2022-07-27T20:15:20.992463Z","shell.execute_reply":"2022-07-27T20:15:30.739899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Config","metadata":{}},{"cell_type":"code","source":"class config:\n    base_dir = \"../input/feedback-prize-effectiveness/\"\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    seed = 42\n    # dataset path \n    train_dataset_path = \"../input/feedbackprizegroupkfolds/train.csv\"\n    test_dataset_path = \"../input/feedbackprizegroupkfolds/test.csv\"\n    sample_submission_path = \"../input/feedbackprizegroupkfolds/sample_submission.csv\"\n       \n    save_dir=\"./result\"\n    \n    #tokenizer params\n    truncation = True \n    padding = 'max_length'\n    max_length = 512\n    \n    # model params\n    model_name = \"microsoft/deberta-v3-base\"\n    \n    #training params\n    learning_rate = 1e-5\n    batch_size = 4\n    epochs = 12\n\nseed_everything(config.seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:30.742985Z","iopub.execute_input":"2022-07-27T20:15:30.743982Z","iopub.status.idle":"2022-07-27T20:15:30.825806Z","shell.execute_reply.started":"2022-07-27T20:15:30.743938Z","shell.execute_reply":"2022-07-27T20:15:30.824334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📊 Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_train_essay(essay_id):\n    parent_path = config.base_dir + 'train'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n\ndef get_test_essay(essay_id):\n    parent_path = config.base_dir + 'test'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:30.827227Z","iopub.execute_input":"2022-07-27T20:15:30.827573Z","iopub.status.idle":"2022-07-27T20:15:30.834359Z","shell.execute_reply.started":"2022-07-27T20:15:30.827537Z","shell.execute_reply":"2022-07-27T20:15:30.833273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:30.837779Z","iopub.execute_input":"2022-07-27T20:15:30.839396Z","iopub.status.idle":"2022-07-27T20:15:30.849395Z","shell.execute_reply.started":"2022-07-27T20:15:30.839358Z","shell.execute_reply":"2022-07-27T20:15:30.848441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(config.train_dataset_path)\ndf_test = pd.read_csv(config.test_dataset_path)\ndf_ss = pd.read_csv(config.sample_submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:30.852107Z","iopub.execute_input":"2022-07-27T20:15:30.852684Z","iopub.status.idle":"2022-07-27T20:15:31.15142Z","shell.execute_reply.started":"2022-07-27T20:15:30.852646Z","shell.execute_reply":"2022-07-27T20:15:31.150476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['essay_text'] = df_train['essay_id'].apply(get_train_essay)\ndf_test['essay_text'] = df_test['essay_id'].apply(get_test_essay)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:31.1527Z","iopub.execute_input":"2022-07-27T20:15:31.15306Z","iopub.status.idle":"2022-07-27T20:15:55.659577Z","shell.execute_reply.started":"2022-07-27T20:15:31.153024Z","shell.execute_reply":"2022-07-27T20:15:55.65863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_mapping = {\n    'Adequate': 0,\n    'Effective': 1,\n    'Ineffective':2\n}\ndf_train['discourse_effectiveness'] = df_train['discourse_effectiveness'].map(target_mapping) ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:55.662761Z","iopub.execute_input":"2022-07-27T20:15:55.663346Z","iopub.status.idle":"2022-07-27T20:15:55.678426Z","shell.execute_reply.started":"2022-07-27T20:15:55.663318Z","shell.execute_reply":"2022-07-27T20:15:55.677575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['discourse_text'] = df_train['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_train['essay_text'] = df_train['essay_text'].apply(resolve_encodings_and_normalize)\n\ndf_test['discourse_text'] = df_test['discourse_text'].apply(resolve_encodings_and_normalize)\ndf_test['essay_text'] = df_test['essay_text'].apply(resolve_encodings_and_normalize)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:15:55.680422Z","iopub.execute_input":"2022-07-27T20:15:55.680796Z","iopub.status.idle":"2022-07-27T20:16:21.197163Z","shell.execute_reply.started":"2022-07-27T20:15:55.680758Z","shell.execute_reply":"2022-07-27T20:16:21.196166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['text'] = df_train['discourse_type'] + \" [SEP] \" + df_train['discourse_text'] + \" [SEP] \" + df_train['essay_text']\ndf_test['text'] = df_test['discourse_type'] + \" [SEP] \" + df_test['discourse_text'] + \" [SEP] \" + df_test['essay_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:21.201945Z","iopub.execute_input":"2022-07-27T20:16:21.202775Z","iopub.status.idle":"2022-07-27T20:16:21.302695Z","shell.execute_reply.started":"2022-07-27T20:16:21.202736Z","shell.execute_reply":"2022-07-27T20:16:21.301484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.model_name)\ntokenizer.save_pretrained(f'{config.save_dir}/tokenizer/')\n\nmodel = AutoModel.from_pretrained(config.model_name)\nmodel.save_pretrained(f'{config.save_dir}/hf_model/')\ndel model \ngc.collect()\ntorch.cuda.empty_cache()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:21.309129Z","iopub.execute_input":"2022-07-27T20:16:21.314773Z","iopub.status.idle":"2022-07-27T20:16:27.238895Z","shell.execute_reply.started":"2022-07-27T20:16:21.314732Z","shell.execute_reply":"2022-07-27T20:16:27.23802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧰 Dataset Prep Function","metadata":{}},{"cell_type":"code","source":"class FeedbackPrizeDataset(Dataset):\n    def __init__(self, text , labels):\n        self.tokenizer = AutoTokenizer.from_pretrained(config.model_name)\n        self.text = text\n        self.label = labels\n    \n    def __len__(self):\n        return len(self.text)\n    \n    def __getitem__(self, idx):\n        text = self.text[idx]\n        label = self.label[idx]\n        train_embeddings = tokenizer(\n            text,truncation = config.truncation,\n            padding = config.padding,\n            max_length = config.max_length, \n        )\n    \n        return {'input_ids': torch.tensor(train_embeddings['input_ids'], dtype = torch.int32),\n                'attention_mask' : torch.tensor(train_embeddings['attention_mask'] , dtype = torch.int32),\n                'target': torch.tensor(label,dtype = torch.long)\n                }\n    \n    def collate(self, batch):\n        dcp = DataCollatorWithPadding(tokenizer=self.tokenizer)\n        return dcp(batch)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:27.24044Z","iopub.execute_input":"2022-07-27T20:16:27.240794Z","iopub.status.idle":"2022-07-27T20:16:27.250917Z","shell.execute_reply.started":"2022-07-27T20:16:27.240759Z","shell.execute_reply":"2022-07-27T20:16:27.250055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔝 Competition Metrics","metadata":{}},{"cell_type":"code","source":"def competition_metrics(y_true, y_preds):\n    return log_loss(y_true, y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:27.252684Z","iopub.execute_input":"2022-07-27T20:16:27.253335Z","iopub.status.idle":"2022-07-27T20:16:27.260084Z","shell.execute_reply.started":"2022-07-27T20:16:27.253292Z","shell.execute_reply":"2022-07-27T20:16:27.25903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧠 Model","metadata":{}},{"cell_type":"code","source":"class FeedbackPrizeModel(pl.LightningModule):\n    def __init__(self, train_dataloader, validation_dataloader):\n        super().__init__()\n        self.transformers_model = AutoModel.from_pretrained(config.model_name)\n        self.classifier =  nn.Linear(self.transformers_model.config.hidden_size,3)\n        self.loss_function = nn.CrossEntropyLoss()\n        self._train_dataloader = train_dataloader\n        self._validation_dataloader = validation_dataloader\n        self.save_hyperparameters()\n        \n    def forward(self, input_ids, attention_mask):\n        cls_token = self.transformers_model(input_ids, attention_mask)[0][:,0,:]\n        output = self.classifier(cls_token)\n        return output\n    \n    def training_step(self,batch,batch_idx):\n        input_ids = batch['input_ids']\n        attention_mask = batch['attention_mask']\n        target = batch['target']\n        output = self(input_ids,attention_mask)\n        loss = self.loss_function(output, target)\n        self.log('train_loss', loss , prog_bar=True)\n        return {'loss': loss}\n    \n    def train_epoch_end(self,outputs):\n        avg_loss = torch.stack([x['loss'] for x in outputs]).mean()\n        print(f'epoch {trainer.current_epoch} training loss {avg_loss}')\n        return {'train_loss': avg_loss} \n    \n    def validation_step(self,batch,batch_idx):\n        input_ids = batch['input_ids']\n        attention_mask = batch['attention_mask']\n        target = batch['target']\n        output = self(input_ids,attention_mask)\n        loss = self.loss_function(output, target)\n        self.log('val_loss', loss , prog_bar=True)\n        return {'val_loss': loss}        \n\n    \n    def validation_epoch_end(self,outputs):\n        avg_loss = torch.stack([x['val_loss'] for x in outputs]).mean()\n        print(f'epoch {trainer.current_epoch} validation loss {avg_loss}')\n        return {'val_loss': avg_loss}\n    \n    def test_step(self):\n        input_ids = batch['input_ids']\n        attention_mask = batch['attention_mask']\n        target = batch['target']\n        output = self(input_ids , attention_mask)\n        loss = self.loss_function(output, target)\n        self.log('test_loss', loss)\n        return {'test_loss': loss}\n    \n    def test_epoch_end(self):\n        avg_loss = torch.stack([x['test_loss'] for x in outputs]).mean()\n        print(f'epoch {trainer.current_epoch} test loss {avg_loss}')\n        return {'test_loss': avg_loss}\n        \n    def train_dataloader(self):\n        return self._train_dataloader \n    \n    def validation_dataloader(self):\n        return self._validation_dataloader\n    \n    def configure_optimizers(self):\n        optimizer = AdamW(self.parameters(), lr = config.learning_rate)\n        return [optimizer]\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:27.261885Z","iopub.execute_input":"2022-07-27T20:16:27.262238Z","iopub.status.idle":"2022-07-27T20:16:27.279402Z","shell.execute_reply.started":"2022-07-27T20:16:27.262203Z","shell.execute_reply":"2022-07-27T20:16:27.278321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(data_loader, model):\n        \n    model.to(config.device)\n    model.eval()\n    model.zero_grad()\n    \n    predictions = []\n    for batch in tqdm(data_loader):\n        inputs = {key:val.reshape(val.shape[0], -1).to(config.device) for key,val in batch.items()}\n        outputs = model(input_ids = inputs['input_ids'], attention_mask = inputs['attention_mask'])\n        outputs = F.softmax(outputs, dim=1)\n        predictions.extend(outputs.detach().cpu().numpy())\n        \n    return predictions","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:27.281964Z","iopub.execute_input":"2022-07-27T20:16:27.282589Z","iopub.status.idle":"2022-07-27T20:16:27.29501Z","shell.execute_reply.started":"2022-07-27T20:16:27.282553Z","shell.execute_reply":"2022-07-27T20:16:27.293982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔄 KFold Training","metadata":{}},{"cell_type":"code","source":"scores = []\nfor fold in range(4,5):\n    print(f\"====== FOLD RUNNING {fold}======\")\n    \n    X_train = df_train.loc[df_train['kfold'] != fold]['text'].tolist()\n    y_train = df_train.loc[df_train['kfold'] != fold]['discourse_effectiveness'].tolist()\n    \n    X_test = df_train.loc[df_train['kfold'] == fold]['text'].tolist()\n    y_test = df_train.loc[df_train['kfold'] == fold]['discourse_effectiveness'].tolist()\n    \n    print(\"Generating Train Dataset\")\n    train_dataset = FeedbackPrizeDataset(X_train,y_train)\n        \n    print(\"Validation Generating Dataset\")\n    validation_dataset = FeedbackPrizeDataset(X_test,y_test)\n\n    print(\"Generating Train DataLoader\")\n    train_dataloader = DataLoader(train_dataset, batch_size = config.batch_size, shuffle = True, num_workers= 2, pin_memory=False)\n    \n    print(\"Generating Validation DataLoader\")\n    validation_dataloader = DataLoader(validation_dataset, batch_size = config.batch_size, shuffle = False, num_workers= 2, pin_memory=False)\n\n    early_stop_callback = EarlyStopping(monitor=\"val_loss\", min_delta=0.00, patience=3, verbose= True, mode=\"min\")\n    checkpoint_callback = ModelCheckpoint(monitor='val_loss',\n                                          dirpath= config.save_dir,\n                                      save_top_k=1,\n                                      save_last= False,\n                                      save_weights_only=True,\n                                      filename= f'./{config.model_name}_{fold}',\n                                      verbose= True,\n                                      mode='min')\n    print(\"Model Creation\")\n    \n    model = FeedbackPrizeModel(train_dataloader, validation_dataloader)\n    \n    trainer = Trainer(max_epochs= config.epochs, gpus = 1, precision=16, accelerator=\"gpu\" ,progress_bar_refresh_rate=10, callbacks=[checkpoint_callback,early_stop_callback])    \n    print(\"Trainer Starting\")\n    trainer.fit(model , train_dataloader , validation_dataloader)  \n\n    print(\"prediction on validation data\")\n    model = FeedbackPrizeModel.load_from_checkpoint(f'{config.save_dir}/{config.model_name}_{fold}.ckpt')\n    preds = predict(validation_dataloader, model)    \n    score = competition_metrics(y_test,preds)\n    scores.append(score)\n    \n    del model,train_dataloader,validation_dataloader,X_train,X_test,y_train,y_test,train_dataset,validation_dataset\n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    print(f\"Log Loss for Fold {fold} is {score}\")\n\nprint(\"the final average Log Loss is \", np.mean(scores))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T20:16:27.296354Z","iopub.execute_input":"2022-07-27T20:16:27.296891Z","iopub.status.idle":"2022-07-28T00:40:58.444975Z","shell.execute_reply.started":"2022-07-27T20:16:27.296852Z","shell.execute_reply":"2022-07-28T00:40:58.443962Z"},"trusted":true},"execution_count":null,"outputs":[]}]}