{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom transformers import DebertaTokenizer, DebertaForSequenceClassification\nfrom sklearn.model_selection import train_test_split\nimport pytorch_lightning as pl\nfrom torch.optim import Adam\nfrom torch.optim.lr_scheduler import StepLR\nfrom pytorch_lightning.callbacks.early_stopping import EarlyStopping\nfrom pytorch_lightning.callbacks import ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:14:23.154005Z","iopub.execute_input":"2022-08-14T22:14:23.154446Z","iopub.status.idle":"2022-08-14T22:14:28.892801Z","shell.execute_reply.started":"2022-08-14T22:14:23.154409Z","shell.execute_reply":"2022-08-14T22:14:28.891533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataSetforTest(Dataset):\n    def __init__(self, ids, texts):\n        self.ids = ids\n        self.texts = texts\n        \n    def __len__(self):\n        return len(self.ids)\n    \n    def __getitem__(self, idx):\n        id = self.ids[idx]\n        text = self.texts[idx]\n        return id, text\n    \ndef collate_fn_for_test(batch):\n    ids, texts = list(zip(*batch))\n    texts = tokenizer(texts, padding=True, return_tensors=\"pt\")\n    return ids, texts\n\nclass Net(pl.LightningModule):\n    def __init__(self, pretrained_model=None):\n        super().__init__()\n        if pretrained_model:\n            print(\"use own model\")\n            self.model = DebertaForSequenceClassification.from_pretrained(pretrained_model, num_labels=3)\n        else:\n            self.model = DebertaForSequenceClassification.from_pretrained(MODEL_NAME, num_labels=3)\n        \n    def forward(self, inputs, labels=None):\n        return self.model(**inputs, labels=labels)\n    \n    def predict(self, outputs):\n        preds = outputs.logits.argmax(axis=1)\n        return preds\n    \n    def training_step(self, batch, batch_idx):\n        ids, inputs, labels = batch\n        loss = self.forward(inputs, labels).loss\n        return loss\n    \n    def validation_step(self, batch, batch_idx):\n        ids, inputs, labels = batch\n        loss = self.forward(inputs, labels).loss\n        self.log(\"val_loss\", loss)\n        return {\"val_loss\": loss}\n    \n    def validation_epo_end(self, outputs):\n        avg_loss = torch.stack([x[\"val_loss\"] for x in outputs]).mean()\n        return {\"avg_val_loss\": avg_loss}\n    \n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=LEARNING_RATE)\n        return [optimizer], StepLR(optimizer, step_size=STEP_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:14:28.895327Z","iopub.execute_input":"2022-08-14T22:14:28.896288Z","iopub.status.idle":"2022-08-14T22:14:28.911109Z","shell.execute_reply.started":"2022-08-14T22:14:28.896244Z","shell.execute_reply":"2022-08-14T22:14:28.909521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 1\ntest_data_path = \"../input/feedback-prize-effectiveness/test.csv\"\ntest_df = pd.read_csv(test_data_path)\ntest_id, test_text = test_df[\"discourse_id\"].tolist(), test_df[\"discourse_text\"].tolist()\ntest_dataset = DataSetforTest(test_id, test_text)\ntest_dataloader = DataLoader(test_dataset, batch_size=BATCH_SIZE, collate_fn=collate_fn_for_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:14:28.913046Z","iopub.execute_input":"2022-08-14T22:14:28.913789Z","iopub.status.idle":"2022-08-14T22:14:28.945472Z","shell.execute_reply.started":"2022-08-14T22:14:28.913743Z","shell.execute_reply":"2022-08-14T22:14:28.944357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls ../input/hg-model/","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:14:28.948144Z","iopub.execute_input":"2022-08-14T22:14:28.949086Z","iopub.status.idle":"2022-08-14T22:14:30.028126Z","shell.execute_reply.started":"2022-08-14T22:14:28.949034Z","shell.execute_reply":"2022-08-14T22:14:30.026858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_CHECKPOINT = \"../input/weights/hg_model/\"\nTOKENIZER_PATH = \"../input/weights/tokeizer/\"\nmodel = Net(pretrained_model=MODEL_CHECKPOINT)\ntokenizer = DebertaTokenizer.from_pretrained(TOKENIZER_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:14:30.030575Z","iopub.execute_input":"2022-08-14T22:14:30.031310Z","iopub.status.idle":"2022-08-14T22:14:36.672781Z","shell.execute_reply.started":"2022-08-14T22:14:30.031268Z","shell.execute_reply":"2022-08-14T22:14:36.671546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.nn import Softmax\nm = Softmax(dim=1)\n## test\npreds = []\nfor batch in test_dataloader:\n    ids, inputs = batch\n    output = model.forward(inputs).logits\n    output = m(output)\n    for i, (discourse_id, logit) in enumerate(zip(ids, output.tolist())):\n        preds.append([discourse_id, *logit])\nfor pred in preds[:]:\n    print(pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:19:30.007945Z","iopub.execute_input":"2022-08-14T22:19:30.008412Z","iopub.status.idle":"2022-08-14T22:19:31.547934Z","shell.execute_reply.started":"2022-08-14T22:19:30.008370Z","shell.execute_reply":"2022-08-14T22:19:31.546634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## save","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndf = pd.DataFrame(preds, columns=[\"discourse_id\", \"Ineffective\", \"Adequate\", \"Effective\"])\ndf.to_csv(\"./submission.csv\", index=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-14T22:19:37.940166Z","iopub.execute_input":"2022-08-14T22:19:37.941285Z","iopub.status.idle":"2022-08-14T22:19:37.971967Z","shell.execute_reply.started":"2022-08-14T22:19:37.941219Z","shell.execute_reply":"2022-08-14T22:19:37.970967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}