{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T20:54:01.189108Z","iopub.execute_input":"2022-07-26T20:54:01.189483Z","iopub.status.idle":"2022-07-26T20:54:01.195886Z","shell.execute_reply.started":"2022-07-26T20:54:01.189441Z","shell.execute_reply":"2022-07-26T20:54:01.194590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!pip install tez","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:01.211223Z","iopub.execute_input":"2022-07-26T20:54:01.212975Z","iopub.status.idle":"2022-07-26T20:54:11.645490Z","shell.execute_reply.started":"2022-07-26T20:54:01.212921Z","shell.execute_reply":"2022-07-26T20:54:11.644164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport torch.nn as nn\nfrom sklearn import metrics, model_selection\nfrom transformers import AutoConfig, AutoModel, AutoTokenizer, get_linear_schedule_with_warmup\nimport numpy as np\nfrom tez import Tez, TezConfig\nfrom tez.callbacks import EarlyStopping\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.648102Z","iopub.execute_input":"2022-07-26T20:54:11.648793Z","iopub.status.idle":"2022-07-26T20:54:11.656262Z","shell.execute_reply.started":"2022-07-26T20:54:11.648745Z","shell.execute_reply":"2022-07-26T20:54:11.655216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.658210Z","iopub.execute_input":"2022-07-26T20:54:11.658618Z","iopub.status.idle":"2022-07-26T20:54:11.810701Z","shell.execute_reply.started":"2022-07-26T20:54:11.658583Z","shell.execute_reply":"2022-07-26T20:54:11.809705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:19:09.030798Z","iopub.execute_input":"2022-07-26T21:19:09.031145Z","iopub.status.idle":"2022-07-26T21:19:09.042269Z","shell.execute_reply.started":"2022-07-26T21:19:09.031116Z","shell.execute_reply":"2022-07-26T21:19:09.041224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:08:38.862434Z","iopub.execute_input":"2022-07-26T21:08:38.863220Z","iopub.status.idle":"2022-07-26T21:08:38.875839Z","shell.execute_reply.started":"2022-07-26T21:08:38.863179Z","shell.execute_reply":"2022-07-26T21:08:38.874839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.824761Z","iopub.execute_input":"2022-07-26T20:54:11.825256Z","iopub.status.idle":"2022-07-26T20:54:11.843224Z","shell.execute_reply.started":"2022-07-26T20:54:11.825166Z","shell.execute_reply":"2022-07-26T20:54:11.842296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModel, AutoModelForSequenceClassification\nfrom transformers import AutoTokenizer","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.844525Z","iopub.execute_input":"2022-07-26T20:54:11.845402Z","iopub.status.idle":"2022-07-26T20:54:11.850319Z","shell.execute_reply.started":"2022-07-26T20:54:11.845366Z","shell.execute_reply":"2022-07-26T20:54:11.849182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n# torch.cuda.is_available()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.852050Z","iopub.execute_input":"2022-07-26T20:54:11.852869Z","iopub.status.idle":"2022-07-26T20:54:11.859890Z","shell.execute_reply.started":"2022-07-26T20:54:11.852812Z","shell.execute_reply":"2022-07-26T20:54:11.859019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# device = \"cuda\"","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.861030Z","iopub.execute_input":"2022-07-26T20:54:11.861652Z","iopub.status.idle":"2022-07-26T20:54:11.868934Z","shell.execute_reply.started":"2022-07-26T20:54:11.861617Z","shell.execute_reply":"2022-07-26T20:54:11.868071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = 'roberta-base'\nmodel = AutoModelForSequenceClassification.from_pretrained(model_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:11.870170Z","iopub.execute_input":"2022-07-26T20:54:11.871058Z","iopub.status.idle":"2022-07-26T20:54:14.298057Z","shell.execute_reply.started":"2022-07-26T20:54:11.871023Z","shell.execute_reply":"2022-07-26T20:54:14.296905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(model_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:14.303592Z","iopub.execute_input":"2022-07-26T20:54:14.304378Z","iopub.status.idle":"2022-07-26T20:54:19.019985Z","shell.execute_reply.started":"2022-07-26T20:54:14.304330Z","shell.execute_reply":"2022-07-26T20:54:19.018692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nclass args:\n    model = \"roberta-base\"\n    epochs = 1\n    batch_size = 32\n    learning_rate = 5e-5\n    train_batch_size = 32\n    valid_batch_size = 32\n    max_len = 128\n    accumulation_steps = 1\n\n\nclass MyDataset:\n    def __init__(self, df, tokenizer, max_len):\n        self.discourse_text = df.discourse_text\n        self.target = df.discourse_effectiveness\n        self.tokenizer = tokenizer\n        self.max_len = max_len\n\n    def __len__(self):\n        return len(self.discourse_text)\n\n    def __getitem__(self, item):\n        text = self.discourse_text[item]\n        inputs = self.tokenizer.encode_plus(\n            text,\n            None,\n            add_special_tokens=True,\n            max_length=self.max_len,\n            padding=\"max_length\",\n            truncation=True,\n        )\n#         print(inputs)\n        ids = inputs[\"input_ids\"]\n        mask = inputs[\"attention_mask\"]\n#         token_type_ids = inputs[\"token_type_ids\"]\n\n        return {\n            \"ids\": torch.tensor(ids, dtype=torch.long),\n            \"mask\": torch.tensor(mask, dtype=torch.long),\n#             \"token_type_ids\": torch.tensor(token_type_ids, dtype=torch.long),\n            \"targets\": torch.tensor(self.target[item], dtype=torch.long),\n        }\n\n\nclass MyModel(nn.Module):\n    def __init__(self, model_name, num_train_steps, learning_rate):\n        super().__init__()\n        self.num_train_steps = num_train_steps\n        self.learning_rate = learning_rate\n        hidden_dropout_prob: float = 0.1\n        layer_norm_eps: float = 1e-7\n\n        config = AutoConfig.from_pretrained(model_name)\n\n        config.update(\n            {\n                \"output_hidden_states\": True,\n                \"hidden_dropout_prob\": hidden_dropout_prob,\n                \"layer_norm_eps\": layer_norm_eps,\n                \"add_pooling_layer\": False,\n                \"num_labels\": 3,\n            }\n        )\n        self.transformer = AutoModelForSequenceClassification.from_pretrained(model_name, config=config)\n#         self.dropout = nn.Dropout(config.hidden_dropout_prob)\n#         self.output = nn.Linear(config.hidden_size, 1)\n\n    def optimizer_scheduler(self):\n        param_optimizer = list(self.named_parameters())\n        no_decay = [\"bias\", \"LayerNorm.weight\"]\n        optimizer_parameters = [\n            {\n                \"params\": [p for n, p in param_optimizer if not any(nd in n for nd in no_decay)],\n                \"weight_decay\": 0.001,\n            },\n            {\n                \"params\": [p for n, p in param_optimizer if any(nd in n for nd in no_decay)],\n                \"weight_decay\": 0.0,\n            },\n        ]\n        opt = torch.optim.AdamW(optimizer_parameters, lr=self.learning_rate)\n        sch = get_linear_schedule_with_warmup(\n            opt,\n            num_warmup_steps=0,\n            num_training_steps=self.num_train_steps,\n        )\n\n        return opt, sch\n\n    def loss(self, outputs, targets):\n        print(outputs)\n#         print(targets)\n        if targets is None:\n            return None\n        return nn.CrossEntropyLoss()(outputs, targets.view(-1))\n\n    def monitor_metrics(self, outputs, targets):\n#         return {}/\n        if targets is None:\n            return {}\n        device = targets.device.type\n#         print(outputs)\n        outputs = np.argmax(torch.sigmoid(outputs).cpu().detach().numpy(), axis=1)\n        print(outputs)\n        targets = targets.cpu().detach().numpy()\n        accuracy = metrics.accuracy_score(targets, outputs)\n        return {\"accuracy\": torch.tensor(accuracy, device=device)}\n\n    def forward(self, ids, mask, targets=None):\n        transformer_out = self.transformer(\n            ids,\n            attention_mask=mask,\n#             token_type_ids=token_type_ids,\n        )\n#         out = transformer_out.pooler_output\n#         out = self.dropout(out)\n#         output = self.output(transformer_out)\n        output = transformer_out.logits\n        loss = self.loss(output, targets)\n        acc = self.monitor_metrics(output, targets)\n        return output, loss, acc","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.021391Z","iopub.execute_input":"2022-07-26T20:54:19.021838Z","iopub.status.idle":"2022-07-26T20:54:19.045045Z","shell.execute_reply.started":"2022-07-26T20:54:19.021792Z","shell.execute_reply":"2022-07-26T20:54:19.043796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_to_idx = {x: i for i,x  in enumerate(set(train_df.discourse_effectiveness))}","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.046635Z","iopub.execute_input":"2022-07-26T20:54:19.047574Z","iopub.status.idle":"2022-07-26T20:54:19.063005Z","shell.execute_reply.started":"2022-07-26T20:54:19.047509Z","shell.execute_reply":"2022-07-26T20:54:19.061947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_to_idx","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.066780Z","iopub.execute_input":"2022-07-26T20:54:19.067098Z","iopub.status.idle":"2022-07-26T20:54:19.079227Z","shell.execute_reply.started":"2022-07-26T20:54:19.067073Z","shell.execute_reply":"2022-07-26T20:54:19.078267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.discourse_effectiveness = train_df.discourse_effectiveness.apply(lambda x: label_to_idx[x])","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.081675Z","iopub.execute_input":"2022-07-26T20:54:19.082462Z","iopub.status.idle":"2022-07-26T20:54:19.111143Z","shell.execute_reply.started":"2022-07-26T20:54:19.082425Z","shell.execute_reply":"2022-07-26T20:54:19.110325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, df_valid = model_selection.train_test_split(\n    train_df, test_size=0.1, random_state=42, stratify=train_df.discourse_effectiveness\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.113203Z","iopub.execute_input":"2022-07-26T20:54:19.113841Z","iopub.status.idle":"2022-07-26T20:54:19.137056Z","shell.execute_reply.started":"2022-07-26T20:54:19.113806Z","shell.execute_reply":"2022-07-26T20:54:19.136244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.reset_index(drop=True)\ndf_valid = df_valid.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.138396Z","iopub.execute_input":"2022-07-26T20:54:19.139374Z","iopub.status.idle":"2022-07-26T20:54:19.146969Z","shell.execute_reply.started":"2022-07-26T20:54:19.139338Z","shell.execute_reply":"2022-07-26T20:54:19.146084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(model_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:19.148493Z","iopub.execute_input":"2022-07-26T20:54:19.149626Z","iopub.status.idle":"2022-07-26T20:54:23.951329Z","shell.execute_reply.started":"2022-07-26T20:54:19.149590Z","shell.execute_reply":"2022-07-26T20:54:23.950280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = MyDataset(\n    df=df_train,\n    tokenizer=tokenizer,\n    max_len=args.max_len,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:23.952913Z","iopub.execute_input":"2022-07-26T20:54:23.953295Z","iopub.status.idle":"2022-07-26T20:54:23.960033Z","shell.execute_reply.started":"2022-07-26T20:54:23.953258Z","shell.execute_reply":"2022-07-26T20:54:23.959111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[2]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:23.961303Z","iopub.execute_input":"2022-07-26T20:54:23.962329Z","iopub.status.idle":"2022-07-26T20:54:23.976368Z","shell.execute_reply.started":"2022-07-26T20:54:23.962292Z","shell.execute_reply":"2022-07-26T20:54:23.975433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset = MyDataset(\n    df=df_valid,\n    tokenizer=tokenizer,\n    max_len=args.max_len,\n)\nvalid_dataset[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:23.977830Z","iopub.execute_input":"2022-07-26T20:54:23.978673Z","iopub.status.idle":"2022-07-26T20:54:23.990099Z","shell.execute_reply.started":"2022-07-26T20:54:23.978634Z","shell.execute_reply":"2022-07-26T20:54:23.988821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_train_steps = int(len(train_dataset) / args.batch_size / args.accumulation_steps * args.epochs)\nprint(n_train_steps)\nmodel = MyModel(\n    model_name=args.model,\n    num_train_steps=n_train_steps,\n    learning_rate=args.learning_rate,\n)\n\nmodel = Tez(model)\n\n\nes = EarlyStopping(monitor=\"valid_loss\", model_path=\"model.bin\")\nconfig = TezConfig(device='gpu',\n    training_batch_size=args.train_batch_size,\n    validation_batch_size=args.valid_batch_size,\n    gradient_accumulation_steps=args.accumulation_steps,\n    epochs=args.epochs,\n    step_scheduler_after=\"batch\",\n)\nmodel.fit(\n    train_dataset,\n    valid_dataset=valid_dataset,\n    callbacks=[es],\n    config=config,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:54:23.991326Z","iopub.execute_input":"2022-07-26T20:54:23.991775Z","iopub.status.idle":"2022-07-26T21:01:17.545209Z","shell.execute_reply.started":"2022-07-26T20:54:23.991739Z","shell.execute_reply":"2022-07-26T21:01:17.543964Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['discourse_effectiveness'] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:19:20.301302Z","iopub.execute_input":"2022-07-26T21:19:20.301689Z","iopub.status.idle":"2022-07-26T21:19:20.307812Z","shell.execute_reply.started":"2022-07-26T21:19:20.301657Z","shell.execute_reply":"2022-07-26T21:19:20.306471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:18:15.083268Z","iopub.execute_input":"2022-07-26T21:18:15.084345Z","iopub.status.idle":"2022-07-26T21:18:15.101880Z","shell.execute_reply.started":"2022-07-26T21:18:15.084290Z","shell.execute_reply":"2022-07-26T21:18:15.100948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = MyDataset(\n    df=test_df,\n    tokenizer=tokenizer,\n    max_len=args.max_len,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:19:30.214070Z","iopub.execute_input":"2022-07-26T21:19:30.214422Z","iopub.status.idle":"2022-07-26T21:19:30.220002Z","shell.execute_reply.started":"2022-07-26T21:19:30.214392Z","shell.execute_reply":"2022-07-26T21:19:30.218592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = torch.nn.Softmax(dim=2)(torch.Tensor(list(model.predict(test_dataset))))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:26:15.873736Z","iopub.execute_input":"2022-07-26T21:26:15.874693Z","iopub.status.idle":"2022-07-26T21:26:16.099938Z","shell.execute_reply.started":"2022-07-26T21:26:15.874654Z","shell.execute_reply":"2022-07-26T21:26:16.098525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_to_idx","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:25:03.824391Z","iopub.execute_input":"2022-07-26T21:25:03.825084Z","iopub.status.idle":"2022-07-26T21:25:03.832589Z","shell.execute_reply.started":"2022-07-26T21:25:03.825035Z","shell.execute_reply":"2022-07-26T21:25:03.831466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ineffectives = []\neffectives = []\nadequates = []\nfor r in results:\n    effectives += r[:,0].tolist()\n    ineffectives += r[:,1].tolist()\n    adequates += r[:,2].tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:32:14.174446Z","iopub.execute_input":"2022-07-26T21:32:14.175171Z","iopub.status.idle":"2022-07-26T21:32:14.180790Z","shell.execute_reply.started":"2022-07-26T21:32:14.175134Z","shell.execute_reply":"2022-07-26T21:32:14.179781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effectives","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:32:22.287709Z","iopub.execute_input":"2022-07-26T21:32:22.288181Z","iopub.status.idle":"2022-07-26T21:32:22.295870Z","shell.execute_reply.started":"2022-07-26T21:32:22.288145Z","shell.execute_reply":"2022-07-26T21:32:22.294770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['discourse_id']","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:25:31.229735Z","iopub.execute_input":"2022-07-26T21:25:31.230087Z","iopub.status.idle":"2022-07-26T21:25:31.237945Z","shell.execute_reply.started":"2022-07-26T21:25:31.230058Z","shell.execute_reply":"2022-07-26T21:25:31.236885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(data={'discourse_id': test_df['discourse_id'], 'Ineffective': ineffectives, 'Adequate': adequates, 'Effective': effectives})","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:35:04.958307Z","iopub.execute_input":"2022-07-26T21:35:04.958932Z","iopub.status.idle":"2022-07-26T21:35:04.967130Z","shell.execute_reply.started":"2022-07-26T21:35:04.958890Z","shell.execute_reply":"2022-07-26T21:35:04.966225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:35:09.093665Z","iopub.execute_input":"2022-07-26T21:35:09.094037Z","iopub.status.idle":"2022-07-26T21:35:09.107552Z","shell.execute_reply.started":"2022-07-26T21:35:09.094008Z","shell.execute_reply":"2022-07-26T21:35:09.106560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('output.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:35:22.549507Z","iopub.execute_input":"2022-07-26T21:35:22.550338Z","iopub.status.idle":"2022-07-26T21:35:22.558783Z","shell.execute_reply.started":"2022-07-26T21:35:22.550297Z","shell.execute_reply":"2022-07-26T21:35:22.557654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}