{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import lightgbm\nimport numpy as np\nimport pandas as pd\nimport os\nfrom typing import Tuple\nimport codecs\nfrom text_unidecode import unidecode\nimport torch\nfrom torch import nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom tqdm import tqdm\nimport torch.nn.functional as F\nfrom transformers import logging\nfrom transformers import AutoModel, AutoTokenizer,AutoConfig\nfrom transformers import RobertaTokenizer, RobertaModel\nlogging.set_verbosity_error()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:35:48.555602Z","iopub.execute_input":"2022-07-27T14:35:48.556538Z","iopub.status.idle":"2022-07-27T14:35:48.563650Z","shell.execute_reply.started":"2022-07-27T14:35:48.556488Z","shell.execute_reply":"2022-07-27T14:35:48.562625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text=pd.read_csv(\"../input/roberta614/new_text.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:35:49.389383Z","iopub.execute_input":"2022-07-27T14:35:49.389974Z","iopub.status.idle":"2022-07-27T14:35:51.486405Z","shell.execute_reply.started":"2022-07-27T14:35:49.389938Z","shell.execute_reply":"2022-07-27T14:35:51.485428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"params1={'objective': 'multiclass',\n        'num_class': 3,\n        'metric':'multi_logloss',\n        'is_unbalance':True,\n        'verbose': -1,\n        'random_state':0,\n        'boosting_type': 'gbdt', 'extra_trees': True, 'reg_alpha': 1.1854083627327756e-06, 'reg_lambda': 3.471705810753103e-05, 'num_leaves': 212, 'colsample_bytree': 0.9545406855115831, 'colsample_bynode': 0.4618904518762159, 'subsample': 0.7876889753751554, 'subsample_freq': 5, 'min_child_samples': 93, 'max_depth': 9, 'n_estimators': 215, 'learning_rate': 0.03131425365648346, 'cat_smooth': 8, 'max_bin': 284}\nparams2={'objective': 'multiclass',\n        'num_class': 3,\n        'metric':'multi_logloss',\n        'is_unbalance':True,\n        'verbose': -1,\n        'random_state':0,\n        'boosting_type': 'dart', 'extra_trees': True, 'reg_alpha': 6.743262365472482e-08, 'reg_lambda': 0.04325551357445395, 'num_leaves': 121, 'colsample_bytree': 0.568564989203252, 'colsample_bynode': 0.4748834898822242, 'subsample': 0.9811330433013815, 'subsample_freq': 5, 'min_child_samples': 36, 'max_depth': 10, 'n_estimators': 292, 'learning_rate': 0.09863647603361064, 'cat_smooth': 28, 'max_bin': 274}\nparams3={'objective': 'multiclass',\n        'num_class': 3,\n        'metric':'multi_logloss',\n        'is_unbalance':True,\n        'verbose': -1,\n        'random_state':0,\n        'boosting_type': 'dart', 'extra_trees': True, 'reg_alpha': 0.02316033692021215, 'reg_lambda': 1.2954787463360906e-07, 'num_leaves': 309, 'colsample_bytree': 0.5952415576068723, 'colsample_bynode': 0.5798053643080928, 'subsample': 0.5943128903197666, 'subsample_freq': 6, 'min_child_samples': 36, 'max_depth': 9, 'n_estimators': 327, 'learning_rate': 0.08602643919880482, 'cat_smooth': 78, 'max_bin': 59}\nparams4={'objective': 'multiclass',\n        'num_class': 3,\n        'metric':'multi_logloss',\n        'is_unbalance':True,\n        'verbose': -1,\n        'random_state':0,\n        'boosting_type': 'gbdt', 'extra_trees': True, 'reg_alpha': 0.017649036299497517, 'reg_lambda': 0.0001837726163182856, 'num_leaves': 177, 'colsample_bytree': 0.8883898323393651, 'colsample_bynode': 0.3774470719302042, 'subsample': 0.9662391229024235, 'subsample_freq': 1, 'min_child_samples': 108, 'max_depth': 7, 'n_estimators': 246, 'learning_rate': 0.036403456457802755, 'cat_smooth': 70, 'max_bin': 114}\nparams5={'objective': 'multiclass',\n        'num_class': 3,\n        'metric':'multi_logloss',\n        'is_unbalance':True,\n        'verbose': -1,\n        'random_state':0,\n        'boosting_type': 'dart', 'extra_trees': True, 'reg_alpha': 0.4446447122218457, 'reg_lambda': 0.4616275827074716, 'num_leaves': 206, 'colsample_bytree': 0.9037754479482342, 'colsample_bynode': 0.5146242269458864, 'subsample': 0.9812691866366688, 'subsample_freq': 1, 'min_child_samples': 102, 'max_depth': 8, 'n_estimators': 380, 'learning_rate': 0.11129085516047715, 'cat_smooth': 23, 'max_bin': 105}\nmodel1=lightgbm.LGBMClassifier(**params1)\nmodel1.fit(datta1,text[\"discourse_effectiveness\"].values)\nmodel2=lightgbm.LGBMClassifier(**params2)\nmodel2.fit(datta2,text[\"discourse_effectiveness\"].values)\nmodel3=lightgbm.LGBMClassifier(**params3)\nmodel3.fit(datta3,text[\"discourse_effectiveness\"].values)\nmodel4=lightgbm.LGBMClassifier(**params4)\nmodel4.fit(datta4,text[\"discourse_effectiveness\"].values)\nmodel5=lightgbm.LGBMClassifier(**params5)\nmodel5.fit(datta5,text[\"discourse_effectiveness\"].values)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:45:17.311367Z","iopub.execute_input":"2022-07-26T13:45:17.311971Z","iopub.status.idle":"2022-07-26T13:59:50.543808Z","shell.execute_reply.started":"2022-07-26T13:45:17.311933Z","shell.execute_reply":"2022-07-26T13:59:50.542644Z"}}},{"cell_type":"code","source":"model1=lightgbm.Booster(model_file=\"../input/roberta614/model1.txt\")\nmodel2=lightgbm.Booster(model_file=\"../input/roberta614/model2.txt\")\nmodel3=lightgbm.Booster(model_file=\"../input/roberta614/model3.txt\")\nmodel4=lightgbm.Booster(model_file=\"../input/roberta614/model4.txt\")\nmodel5=lightgbm.Booster(model_file=\"../input/roberta614/model5.txt\")\nmodels1=[model1,model2,model3,model4,model5]\nmodel6=lightgbm.Booster(model_file=\"../input/debertav3large/deberta_model0.txt\")\nmodel7=lightgbm.Booster(model_file=\"../input/debertav3large/deberta_model1.txt\")\nmodel8=lightgbm.Booster(model_file=\"../input/debertav3large/deberta_model2.txt\")\nmodel9=lightgbm.Booster(model_file=\"../input/debertav3large/deberta_model3.txt\")\nmodel10=lightgbm.Booster(model_file=\"../input/debertav3large/deberta_model4.txt\")\nmodels2=[model6,model7,model8,model9,model10]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:35:54.355984Z","iopub.execute_input":"2022-07-27T14:35:54.356678Z","iopub.status.idle":"2022-07-27T14:35:54.787391Z","shell.execute_reply.started":"2022-07-27T14:35:54.356637Z","shell.execute_reply":"2022-07-27T14:35:54.786321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = RobertaTokenizer.from_pretrained('../input/my-roberta-base')\ntokenizer.model_max_length=512\nclass FeedBackModel(nn.Module):\n    def __init__(self):\n        super(FeedBackModel, self).__init__()\n        self.model = RobertaModel.from_pretrained('../input/my-roberta-base')\n        self.linear = nn.Linear(768, 3)\n\n    def forward(self, inputs):\n        last_hidden_states = self.model(**inputs)[0][:, 0, :]\n        return last_hidden_states\nmodell=FeedBackModel()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:35:56.651382Z","iopub.execute_input":"2022-07-27T14:35:56.652048Z","iopub.status.idle":"2022-07-27T14:36:04.000667Z","shell.execute_reply.started":"2022-07-27T14:35:56.652014Z","shell.execute_reply":"2022-07-27T14:36:03.999670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay(essay_id, is_train=True):\n    parent_path = INPUT_DIR + 'train' if is_train else INPUT_DIR + 'test'\n    essay_path = os.path.join(parent_path, f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n\nINPUT_DIR = \"../input/feedback-prize-effectiveness/\"\ntext = pd.read_csv(os.path.join(INPUT_DIR, 'test.csv'))\ntext['essay_text']  = text['essay_id'].apply(lambda x: get_essay(x, is_train=False))\n\ndef replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start: error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start: error.end].decode(\"cp1252\"), error.end\n\n\n# Register the encoding and decoding error handlers for `utf-8` and `cp1252`.\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    \"\"\"Resolve the encoding problems and normalize the abnormal characters.\"\"\"\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    text = unidecode(text)\n    return text\n\ntext['composed_text']=text['discourse_type'].str.lower().str.strip() + \" \" \\\n                + text['discourse_text'].str.lower().str.strip()\ntext['essay_text']=text['essay_text'].str.lower().str.strip()\ntext['discourse_type']=text[\"discourse_type\"].astype('category').cat.codes","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:04.002982Z","iopub.execute_input":"2022-07-27T14:36:04.003489Z","iopub.status.idle":"2022-07-27T14:36:04.048018Z","shell.execute_reply.started":"2022-07-27T14:36:04.003450Z","shell.execute_reply":"2022-07-27T14:36:04.047172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Feedback_Data(Dataset):\n    def __init__(self,df):\n        self.df=df\n    def __getitem__(self, idx):\n        composed_text,composed_text_mask=tokenizer(self.df.iloc[idx][\"composed_text\"], self.df.iloc[idx][\"essay_text\"], add_special_tokens=True, padding='max_length',truncation=True, return_tensors='pt').values()\n        return composed_text,composed_text_mask\n    def __len__(self):\n        return len(self.df)\n    \ntest_loader = DataLoader(Feedback_Data(text),\n                          batch_size=1,\n                          shuffle=False,\n                          num_workers=0, pin_memory=False, drop_last=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:04.049503Z","iopub.execute_input":"2022-07-27T14:36:04.049859Z","iopub.status.idle":"2022-07-27T14:36:04.057063Z","shell.execute_reply.started":"2022-07-27T14:36:04.049826Z","shell.execute_reply":"2022-07-27T14:36:04.056058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.DataFrame(np.zeros((len(text),3)),columns=['Adequate','Effective','Ineffective'])\nfor i in range(1,6):\n    modell.load_state_dict(torch.load('../input/roberta614/roberta1{:}.pt'.format(i)))\n    modell.cuda()\n    datta_test=[]\n    with tqdm(test_loader) as t:\n        for data in t:\n            with torch.no_grad():\n                composed_dict={\"input_ids\":data[0].squeeze(1).cuda(),\"attention_mask\":data[1].squeeze(1).cuda()}\n                datta_test.append(modell(composed_dict).cpu().numpy())\n    datta_test=np.concatenate(datta_test,axis=0)\n    submission+=pd.DataFrame(models1[i-1].predict(datta_test),columns=['Adequate','Effective','Ineffective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:04.059958Z","iopub.execute_input":"2022-07-27T14:36:04.060381Z","iopub.status.idle":"2022-07-27T14:36:07.495022Z","shell.execute_reply.started":"2022-07-27T14:36:04.060346Z","shell.execute_reply":"2022-07-27T14:36:07.493388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer=AutoTokenizer.from_pretrained(\"../input/debertav3large/deberta/tokenizer/\")\ntokenizer.model_max_length=512\nclass MeanPooling(nn.Module):\n    def __init__(self):\n        super(MeanPooling, self).__init__()\n\n    def forward(self, last_hidden_state, attention_mask):\n        input_mask_expanded = attention_mask.unsqueeze(-1).expand(last_hidden_state.size()).float()\n        sum_embeddings = torch.sum(last_hidden_state * input_mask_expanded, 1)\n        sum_mask = input_mask_expanded.sum(1)\n        sum_mask = torch.clamp(sum_mask, min=1e-9) #\n        mean_embeddings = sum_embeddings / sum_mask\n        return mean_embeddings\n\nclass FeedBackModel(nn.Module):\n    def __init__(self):\n        super(FeedBackModel, self).__init__()\n        # Header (fast or normal)\n        self.model = AutoModel.from_pretrained(\"../input/deberta-v3-base/deberta-v3-base\")\n\n        self.config = AutoConfig.from_pretrained(\"../input/deberta-v3-base/deberta-v3-base\")\n        self.drop = nn.Dropout(p=0.1)\n        self.pooler = MeanPooling()\n        self.fc = nn.Linear(self.config.hidden_size, 3)\n\n    def forward(self, ids, mask):\n        out = self.model(input_ids=ids,\n                         attention_mask=mask,\n                         output_hidden_states=False)\n        out = self.pooler(out.last_hidden_state, mask)\n        return out\nmodell=FeedBackModel()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:07.497026Z","iopub.execute_input":"2022-07-27T14:36:07.497475Z","iopub.status.idle":"2022-07-27T14:36:10.325875Z","shell.execute_reply.started":"2022-07-27T14:36:07.497438Z","shell.execute_reply":"2022-07-27T14:36:10.324061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Feedback_Data(Dataset):\n    def __init__(self,df):\n        self.df=df\n    def __getitem__(self, idx):\n        composed_text,_,composed_text_mask=tokenizer(self.df.iloc[idx][\"composed_text\"], self.df.iloc[idx][\"essay_text\"], add_special_tokens=True, padding='max_length',truncation=True, max_length=512, return_tensors='pt').values()\n        return composed_text,composed_text_mask\n    def __len__(self):\n        return len(self.df)\n    \ntest_loader = DataLoader(Feedback_Data(text),\n                          batch_size=1,\n                          shuffle=False,\n                          num_workers=0, pin_memory=False, drop_last=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:10.328107Z","iopub.execute_input":"2022-07-27T14:36:10.328847Z","iopub.status.idle":"2022-07-27T14:36:10.338829Z","shell.execute_reply.started":"2022-07-27T14:36:10.328795Z","shell.execute_reply":"2022-07-27T14:36:10.337777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(5):\n    modell.load_state_dict(torch.load('../input/debertav3large/deberta/models-deberta-v3-base-deberta-v3-base_fold{:}_best.pth'.format(i)))\n    modell.cuda()\n    datta_test=[]\n    with tqdm(test_loader) as t:\n        for data in t:\n            with torch.no_grad():\n                datta_test.append(modell(data[0].squeeze(1).cuda(),data[1].squeeze(1).cuda()).cpu().numpy())\n    datta_test=np.concatenate(datta_test,axis=0)\n    submission+=pd.DataFrame(models2[i].predict(datta_test),columns=['Adequate','Effective','Ineffective'])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:10.341523Z","iopub.execute_input":"2022-07-27T14:36:10.341951Z","iopub.status.idle":"2022-07-27T14:36:29.139912Z","shell.execute_reply.started":"2022-07-27T14:36:10.341913Z","shell.execute_reply":"2022-07-27T14:36:29.138886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission/=10\nsubmission[\"discourse_id\"]=text[\"discourse_id\"]\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T14:36:29.141314Z","iopub.execute_input":"2022-07-27T14:36:29.141763Z","iopub.status.idle":"2022-07-27T14:36:29.167043Z","shell.execute_reply.started":"2022-07-27T14:36:29.141724Z","shell.execute_reply":"2022-07-27T14:36:29.166185Z"},"trusted":true},"execution_count":null,"outputs":[]}]}