{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing required libraries","metadata":{}},{"cell_type":"markdown","source":"Some of this code is borrowed from [https://www.kaggle.com/code/sapeksh/ensemble-deberta-roberta-lgbm](http://) please consider upvoting this notebook","metadata":{}},{"cell_type":"markdown","source":"# If you find this notebook useful please upvote this notebook ","metadata":{}},{"cell_type":"code","source":"import gc\nimport os \nimport pickle\nimport glob\n\nfrom text_unidecode import unidecode\nfrom typing import Dict,List,Tuple\nimport codecs\n\n# manupulaters\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\nimport seaborn as sns\n# torch \nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn import Parameter\nfrom torch.utils.data import Dataset,DataLoader\n\n# transformers\nfrom transformers import AutoModel,AutoTokenizer,AutoConfig\n\n#warnings\nimport warnings\nwarnings.simplefilter(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:35.972234Z","iopub.execute_input":"2022-08-11T05:05:35.972672Z","iopub.status.idle":"2022-08-11T05:05:38.966600Z","shell.execute_reply.started":"2022-08-11T05:05:35.972636Z","shell.execute_reply":"2022-08-11T05:05:38.965657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    \n    text = unidecode(text)\n    \n    return text\n\n\ndef fetch_essay(essay_id: str, txt_dir: str):\n    essay_path = os.path.join(COMP_DIR + txt_dir, essay_id + '.txt')\n    essay_text = open(essay_path, 'r').read()\n    \n    return essay_text\n\n\ndef prepare_input(cfg, text, text_2=None):\n    inputs = cfg.tokenizer(text, text_2,\n                           padding=\"max_length\",\n                           add_special_tokens=True,\n                           max_length=cfg.max_len,\n                           truncation=True)\n\n    for k, v in inputs.items():\n        inputs[k] = torch.tensor(v, dtype=torch.long)\n        \n    return inputs\n\n\ndef inference_fn(test_loader, model, device):\n    preds = []\n    model.eval()\n    model.to(device)\n    tk0 = tqdm(test_loader, total=len(test_loader))\n    for inputs in tk0:\n        for k, v in inputs.items():\n            inputs[k] = v.to(device)\n        with torch.no_grad():\n            output = model(inputs)\n        \n        preds.append(F.softmax(output).to('cpu').numpy())\n\n    return np.concatenate(preds)  \n\ndef show_gradient(df, n_row=None):\n    if not n_row:\n        n_row = 5\n\n    return df.head(n_row) \\\n                .assign(all_mean=lambda x: x.mean(axis=1)) \\\n                    .style.background_gradient(cmap=cm, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:38.968382Z","iopub.execute_input":"2022-08-11T05:05:38.968932Z","iopub.status.idle":"2022-08-11T05:05:38.983541Z","shell.execute_reply.started":"2022-08-11T05:05:38.968875Z","shell.execute_reply":"2022-08-11T05:05:38.982357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.precision', 4)\ncm = sns.light_palette('green', as_cmap=True)\nprops_param = \"color:white; font-weight:bold; background-color:green;\"\n\nN_ROW = 10\n\nCOMP_DIR = \"../input/feedback-prize-effectiveness/\"\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:38.985084Z","iopub.execute_input":"2022-08-11T05:05:38.986120Z","iopub.status.idle":"2022-08-11T05:05:39.060293Z","shell.execute_reply.started":"2022-08-11T05:05:38.986084Z","shell.execute_reply":"2022-08-11T05:05:39.059170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = COMP_DIR + \"test.csv\"\nsubmission_path = COMP_DIR + \"sample_submission.csv\"\n\ntest_origin = pd.read_csv(test_path)\nsubmission_origin = pd.read_csv(submission_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.063305Z","iopub.execute_input":"2022-08-11T05:05:39.064011Z","iopub.status.idle":"2022-08-11T05:05:39.086767Z","shell.execute_reply.started":"2022-08-11T05:05:39.063957Z","shell.execute_reply":"2022-08-11T05:05:39.085935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_origin.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.090006Z","iopub.execute_input":"2022-08-11T05:05:39.091836Z","iopub.status.idle":"2022-08-11T05:05:39.110764Z","shell.execute_reply.started":"2022-08-11T05:05:39.091805Z","shell.execute_reply":"2022-08-11T05:05:39.109863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = \"../input/feedback-prize-effectiveness/train.csv\"\ncols_list = ['essay_id', 'discourse_text']\nidxs_list = [49, 80, 945, 947, 1870]\n\ntemp = pd.read_csv(data_path, usecols=cols_list).loc[idxs_list, :]\ntemp","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.112312Z","iopub.execute_input":"2022-08-11T05:05:39.112940Z","iopub.status.idle":"2022-08-11T05:05:39.398867Z","shell.execute_reply.started":"2022-08-11T05:05:39.112882Z","shell.execute_reply":"2022-08-11T05:05:39.397822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['discourse_text_resolved'] = temp['discourse_text'].apply(resolve_encodings_and_normalize)\n\ntemp['essay_text'] = temp['essay_id'].transform(fetch_essay, txt_dir='train')\ntemp['essay_text_resolved'] = temp['essay_text'].apply(resolve_encodings_and_normalize)\n\n#temp.drop(['discourse_text_UPD','essay_text_UPD'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.400747Z","iopub.execute_input":"2022-08-11T05:05:39.401108Z","iopub.status.idle":"2022-08-11T05:05:39.433856Z","shell.execute_reply.started":"2022-08-11T05:05:39.401072Z","shell.execute_reply":"2022-08-11T05:05:39.433034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for n, row in enumerate(temp.iterrows()):\n    indx, data = row\n    disc_text = data.discourse_text\n    disc_text_upd = data.discourse_text_resolved\n\n    print(f'\\nN{n} === index: {indx} ===')\n    print(f'\\n>>> original text:')\n    print(repr(disc_text))\n    print(f'\\n>>> resolved text:')\n    print(repr(disc_text_upd))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.435772Z","iopub.execute_input":"2022-08-11T05:05:39.436385Z","iopub.status.idle":"2022-08-11T05:05:39.446952Z","shell.execute_reply.started":"2022-08-11T05:05:39.436349Z","shell.execute_reply":"2022-08-11T05:05:39.445919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions **DeBertA**","metadata":{}},{"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, cfg, df):\n        self.cfg = cfg\n        self.text = df['text'].values\n\n    def __len__(self):\n        return len(self.text)\n\n    def __getitem__(self, item):    \n        text = self.text[item]\n        inputs = prepare_input(self.cfg, text)\n        \n        return inputs\n\nclass CustomModel(nn.Module):\n    def __init__(self, cfg, config_path=None, pretrained=False):\n        super().__init__()\n        self.cfg = cfg\n        \n        if config_path is None:\n            self.config = AutoConfig.from_pretrained(cfg.model, output_hidden_states=True)\n        else:\n            self.config = torch.load(config_path)\n        \n        if pretrained:\n            self.model = AutoModel.from_pretrained(cfg.model, config=self.config)\n        else:\n            self.model = AutoModel.from_config(self.config)\n        \n        self.bilstm = nn.LSTM(self.config.hidden_size, (self.config.hidden_size) // 2, num_layers=2, \n                              dropout=self.config.hidden_dropout_prob, batch_first=True,\n                              bidirectional=True)\n        \n        # self.dropout = nn.Dropout(0.2)\n        self.dropout1 = nn.Dropout(0.1)\n        self.dropout2 = nn.Dropout(0.2)\n        self.dropout3 = nn.Dropout(0.2)\n        self.dropout4 = nn.Dropout(0.3)\n        self.dropout5 = nn.Dropout(0.4)\n        \n        self.output = nn.Sequential(\n            nn.Linear(self.config.hidden_size, 3)  # self.cfg.target_size\n        )\n                \n    def _init_weights(self, module):\n        if isinstance(module, nn.Linear):\n            module.weight.data.normal_(mean=0.0, std=self.config.initializer_range)\n            if module.bias is not None:\n                module.bias.data.zero_()\n        elif isinstance(module, nn.Embedding):\n            module.weight.data.normal_(mean=0.0, std=self.config.initializer_range)\n            if module.padding_idx is not None:\n                module.weight.data[module.padding_idx].zero_()\n        elif isinstance(module, nn.LayerNorm):\n            module.bias.data.zero_()\n            module.weight.data.fill_(1.0)\n\n    def forward(self, inputs):\n        sequence_output = self.model(**inputs)[0][:, 0, :]\n\n        logits1 = self.output(self.dropout1(sequence_output))\n        logits2 = self.output(self.dropout2(sequence_output))\n        logits3 = self.output(self.dropout3(sequence_output))\n        logits4 = self.output(self.dropout4(sequence_output))\n        logits5 = self.output(self.dropout5(sequence_output))\n        logits = (logits1 + logits2 + logits3 + logits4 + logits5) / 5\n\n        return logits\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.450409Z","iopub.execute_input":"2022-08-11T05:05:39.451267Z","iopub.status.idle":"2022-08-11T05:05:39.467743Z","shell.execute_reply.started":"2022-08-11T05:05:39.451238Z","shell.execute_reply":"2022-08-11T05:05:39.466621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    path = \"../input/feedback-deberta-large-051/\"\n    config_path = path+'config.pth'\n    model = \"microsoft/deberta-large\"\n    num_workers = 2\n    batch_size = 16\n    max_len = 512\n    seed = 42\n    n_fold = 4\n    # trn_fold = [0, 1, 2, 3]\n    # fc_dropout = 0.2\n    # target_size = 3\n    \nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.path + 'tokenizer')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:39.620343Z","iopub.execute_input":"2022-08-11T05:05:39.621001Z","iopub.status.idle":"2022-08-11T05:05:39.789271Z","shell.execute_reply.started":"2022-08-11T05:05:39.620961Z","shell.execute_reply":"2022-08-11T05:05:39.788204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test_origin.copy()\nSEP = CFG.tokenizer.sep_token\n\ndf['discourse_text'] = df['discourse_text'].apply(resolve_encodings_and_normalize)\ndf['essay_text'] = df['essay_id'].transform(fetch_essay, txt_dir='test')\ndf['essay_text'] = df['essay_text'].apply(resolve_encodings_and_normalize)\ndf['text'] = df['discourse_type'] + ' ' + df['discourse_text'] + SEP + df['essay_text']\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:40.066312Z","iopub.execute_input":"2022-08-11T05:05:40.067250Z","iopub.status.idle":"2022-08-11T05:05:40.104288Z","shell.execute_reply.started":"2022-08-11T05:05:40.067200Z","shell.execute_reply":"2022-08-11T05:05:40.103356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = TestDataset(CFG, df)\ntest_loader = DataLoader(test_dataset,\n                         batch_size=CFG.batch_size,\n                         shuffle=False,\n                         num_workers=CFG.num_workers,\n                         pin_memory=True, drop_last=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:40.395975Z","iopub.execute_input":"2022-08-11T05:05:40.396623Z","iopub.status.idle":"2022-08-11T05:05:40.404771Z","shell.execute_reply.started":"2022-08-11T05:05:40.396589Z","shell.execute_reply":"2022-08-11T05:05:40.403804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deberta_predictions = []\n\nfor fold in range(CFG.n_fold):\n    model = CustomModel(CFG, config_path=CFG.config_path, pretrained=False)\n    state = torch.load(CFG.path+f\"{CFG.model.replace('/', '-')}_fold{fold}_best.pth\",\n                       map_location=torch.device('cpu'))\n    \n    model.load_state_dict(state['model'])\n    prediction = inference_fn(test_loader, model, DEVICE)\n    \n    deberta_predictions.append(prediction)\n    \n    del model, state, prediction; gc.collect()\n    torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:05:40.682661Z","iopub.execute_input":"2022-08-11T05:05:40.683372Z","iopub.status.idle":"2022-08-11T05:07:32.684130Z","shell.execute_reply.started":"2022-08-11T05:05:40.683336Z","shell.execute_reply":"2022-08-11T05:07:32.683049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deb_ineffective = []\ndeb_effective = []\ndeb_adequate = []\n\nfor x in deberta_predictions:\n    deb_ineffective.append(x[:, 0])\n    deb_adequate.append(x[:, 1])\n    deb_effective.append(x[:, 2])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.763941Z","iopub.execute_input":"2022-08-11T05:07:32.764384Z","iopub.status.idle":"2022-08-11T05:07:32.772498Z","shell.execute_reply.started":"2022-08-11T05:07:32.764342Z","shell.execute_reply":"2022-08-11T05:07:32.771551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deb_ineffective = pd.DataFrame(deb_ineffective).T\n\nshow_gradient(\n    deb_ineffective,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.774738Z","iopub.execute_input":"2022-08-11T05:07:32.775137Z","iopub.status.idle":"2022-08-11T05:07:32.859679Z","shell.execute_reply.started":"2022-08-11T05:07:32.775100Z","shell.execute_reply":"2022-08-11T05:07:32.858597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deb_adequate = pd.DataFrame(deb_adequate).T\n\nshow_gradient(\n    deb_adequate,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.862606Z","iopub.execute_input":"2022-08-11T05:07:32.862988Z","iopub.status.idle":"2022-08-11T05:07:32.889862Z","shell.execute_reply.started":"2022-08-11T05:07:32.862950Z","shell.execute_reply":"2022-08-11T05:07:32.888773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deb_effective = pd.DataFrame(deb_effective).T\n\nshow_gradient(\n    deb_effective,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.891570Z","iopub.execute_input":"2022-08-11T05:07:32.891953Z","iopub.status.idle":"2022-08-11T05:07:32.918850Z","shell.execute_reply.started":"2022-08-11T05:07:32.891914Z","shell.execute_reply":"2022-08-11T05:07:32.917939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Roberta","metadata":{}},{"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, cfg, df):\n        self.cfg = cfg\n        self.discourse = df['discourse'].values\n        self.essay = df['essay'].values\n        \n    def __len__(self):\n        return len(self.discourse)\n    \n    def __getitem__(self, item):\n        discourse = self.discourse[item]\n        essay = self.essay[item]\n        \n        inputs = prepare_input(self.cfg, discourse, essay)\n        \n        return inputs\n        \nclass FeedBackModel(nn.Module):\n    def __init__(self, model_path):\n        super(FeedBackModel, self).__init__()\n        self.model = AutoModel.from_pretrained(model_path)\n        self.linear = nn.Linear(768, 3)\n\n    def forward(self, inputs):\n        last_hidden_states = self.model(**inputs)[0][:, 0, :]\n        outputs = self.linear(last_hidden_states)\n        \n        return outputs","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.920858Z","iopub.execute_input":"2022-08-11T05:07:32.921445Z","iopub.status.idle":"2022-08-11T05:07:32.930962Z","shell.execute_reply.started":"2022-08-11T05:07:32.921408Z","shell.execute_reply":"2022-08-11T05:07:32.930050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_list = pickle.load(\n    open(\"../input/feedback-roberta-ep1/roberta_modellist_ep2.pkl\", \"rb\")\n)\n\nclass CFG:\n    path = \"../input/roberta-base/\"\n    n_fold = 5\n    batch = 16\n    max_len = 512\n    num_workers = 2\n    \nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.path)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:32.933475Z","iopub.execute_input":"2022-08-11T05:07:32.933736Z","iopub.status.idle":"2022-08-11T05:07:57.949115Z","shell.execute_reply.started":"2022-08-11T05:07:32.933712Z","shell.execute_reply":"2022-08-11T05:07:57.948146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test_origin.copy()\n\ntxt_sep = \" \"\ndf['discourse'] = df['discourse_type'].str.lower().str.strip() + txt_sep \\\n                + df['discourse_text'].str.lower().str.strip()\n\ndf['essay'] = df['essay_id'].transform(fetch_essay, txt_dir='test').str.lower().str.strip()\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:57.950758Z","iopub.execute_input":"2022-08-11T05:07:57.951127Z","iopub.status.idle":"2022-08-11T05:07:57.977765Z","shell.execute_reply.started":"2022-08-11T05:07:57.951091Z","shell.execute_reply":"2022-08-11T05:07:57.976875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = TestDataset(CFG, df)\ntest_loader = DataLoader(test_dataset, batch_size=CFG.batch,\n                         shuffle=False, num_workers=CFG.num_workers,\n                         pin_memory=True, drop_last=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:57.978974Z","iopub.execute_input":"2022-08-11T05:07:57.979321Z","iopub.status.idle":"2022-08-11T05:07:57.986793Z","shell.execute_reply.started":"2022-08-11T05:07:57.979286Z","shell.execute_reply":"2022-08-11T05:07:57.985854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roberta_predicts = []\nfor i in range(CFG.n_fold):\n    model = model_list[i]\n    \n    prediction = inference_fn(test_loader, model, DEVICE)\n    roberta_predicts.append(prediction)\n    \n    del model, prediction\n    torch.cuda.empty_cache()    \n    gc.collect()\n    \ndel model_list\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:07:57.992331Z","iopub.execute_input":"2022-08-11T05:07:57.992601Z","iopub.status.idle":"2022-08-11T05:08:01.146777Z","shell.execute_reply.started":"2022-08-11T05:07:57.992576Z","shell.execute_reply":"2022-08-11T05:08:01.145680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rob_ineffective = []\nrob_effective = []\nrob_adequate = []\n\nfor x in roberta_predicts:\n    rob_ineffective.append(x[:, 0])\n    rob_adequate.append(x[:, 1])\n    rob_effective.append(x[:, 2])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.148394Z","iopub.execute_input":"2022-08-11T05:08:01.149016Z","iopub.status.idle":"2022-08-11T05:08:01.157489Z","shell.execute_reply.started":"2022-08-11T05:08:01.148954Z","shell.execute_reply":"2022-08-11T05:08:01.156550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rob_ineffective = pd.DataFrame(rob_ineffective).T\n\nshow_gradient(\n    rob_ineffective,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.159391Z","iopub.execute_input":"2022-08-11T05:08:01.159770Z","iopub.status.idle":"2022-08-11T05:08:01.192907Z","shell.execute_reply.started":"2022-08-11T05:08:01.159730Z","shell.execute_reply":"2022-08-11T05:08:01.191900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rob_adequate = pd.DataFrame(rob_adequate).T\n\nshow_gradient(\n    rob_adequate,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.194277Z","iopub.execute_input":"2022-08-11T05:08:01.194712Z","iopub.status.idle":"2022-08-11T05:08:01.222031Z","shell.execute_reply.started":"2022-08-11T05:08:01.194674Z","shell.execute_reply":"2022-08-11T05:08:01.221072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rob_effective = pd.DataFrame(rob_effective).T\n\nshow_gradient(\n    rob_effective,\n    N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.223554Z","iopub.execute_input":"2022-08-11T05:08:01.223925Z","iopub.status.idle":"2022-08-11T05:08:01.252223Z","shell.execute_reply.started":"2022-08-11T05:08:01.223873Z","shell.execute_reply":"2022-08-11T05:08:01.251309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nxg=pd.read_csv('../input/feedback-prize-evaluating-longformer-lgbm/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.253756Z","iopub.execute_input":"2022-08-11T05:08:01.254415Z","iopub.status.idle":"2022-08-11T05:08:01.267660Z","shell.execute_reply.started":"2022-08-11T05:08:01.254378Z","shell.execute_reply":"2022-08-11T05:08:01.266763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"bert=pd.read_csv('../input/bertinference/submission (6).csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:34:57.876400Z","iopub.execute_input":"2022-08-09T13:34:57.876710Z","iopub.status.idle":"2022-08-09T13:34:57.891645Z","shell.execute_reply.started":"2022-08-09T13:34:57.876682Z","shell.execute_reply":"2022-08-09T13:34:57.890652Z"}}},{"cell_type":"code","source":"xg.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.269160Z","iopub.execute_input":"2022-08-11T05:08:01.269570Z","iopub.status.idle":"2022-08-11T05:08:01.280668Z","shell.execute_reply.started":"2022-08-11T05:08:01.269533Z","shell.execute_reply":"2022-08-11T05:08:01.279583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_ineffective=[]\nxg_adequate=[]\nxg_effective=[]\n\nfor i in xg['Ineffective']:\n    xg_ineffective.append(i)\nfor i in xg['Adequate']:\n    xg_adequate.append(i)\nfor i in xg['Effective']:\n    xg_effective.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.282402Z","iopub.execute_input":"2022-08-11T05:08:01.282906Z","iopub.status.idle":"2022-08-11T05:08:01.291168Z","shell.execute_reply.started":"2022-08-11T05:08:01.282854Z","shell.execute_reply":"2022-08-11T05:08:01.290074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_in = pd.DataFrame(xg_ineffective)\nxg_ad = pd.DataFrame(xg_adequate)\nxg_ef = pd.DataFrame(xg_effective)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.292590Z","iopub.execute_input":"2022-08-11T05:08:01.293561Z","iopub.status.idle":"2022-08-11T05:08:01.300466Z","shell.execute_reply.started":"2022-08-11T05:08:01.293525Z","shell.execute_reply":"2022-08-11T05:08:01.299412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ber_ineffective=[]\nber_adequate=[]\nber_effective=[]\n\nfor i in bert['Ineffective']:\n    ber_ineffective.append(i)\nfor i in bert['Adequate']:\n    ber_adequate.append(i)\nfor i in bert['Effective']:\n    ber_effective.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:34:57.937122Z","iopub.execute_input":"2022-08-09T13:34:57.937483Z","iopub.status.idle":"2022-08-09T13:34:57.946476Z","shell.execute_reply.started":"2022-08-09T13:34:57.937454Z","shell.execute_reply":"2022-08-09T13:34:57.945127Z"}}},{"cell_type":"markdown","source":"be_in = pd.DataFrame(ber_ineffective)\nbe_ad = pd.DataFrame(ber_adequate)\nbe_ef = pd.DataFrame(ber_effective)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:34:57.949070Z","iopub.execute_input":"2022-08-09T13:34:57.950185Z","iopub.status.idle":"2022-08-09T13:34:57.957863Z","shell.execute_reply.started":"2022-08-09T13:34:57.950154Z","shell.execute_reply":"2022-08-09T13:34:57.956762Z"}}},{"cell_type":"code","source":"xg_in","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:01.302123Z","iopub.execute_input":"2022-08-11T05:08:01.302739Z","iopub.status.idle":"2022-08-11T05:08:01.315492Z","shell.execute_reply.started":"2022-08-11T05:08:01.302694Z","shell.execute_reply":"2022-08-11T05:08:01.314455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import log_loss\nimport gensim\nfrom scipy import sparse\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:08:58.611273Z","iopub.execute_input":"2022-08-11T05:08:58.612302Z","iopub.status.idle":"2022-08-11T05:09:00.784788Z","shell.execute_reply.started":"2022-08-11T05:08:58.612252Z","shell.execute_reply":"2022-08-11T05:09:00.783803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    n_folds = 4\n    \nINPUT_DIR = \"../input/feedback-prize-effectiveness/\"\n\ndef get_train_essay(essay_id):\n    essay_path = os.path.join(INPUT_DIR,f'train/{essay_id}.txt')\n    essay_text = open(essay_path,'r').read()\n    return essay_text\n\ndef get_test_essay(essay_id):\n    essay_path = os.path.join(INPUT_DIR,f'test/{essay_id}.txt')\n    essay_text = open(essay_path,'r').read()\n    return essay_text\n\ntrain = pd.read_csv(INPUT_DIR+'train.csv')\ntest = pd.read_csv(INPUT_DIR+'test.csv')\ntrain['essay_text'] = train['essay_id'].apply(get_train_essay)\ntest['essay_text'] = test['essay_id'].apply(get_test_essay)\n\ndef set_seed(seed=42):\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nset_seed(CFG.seed)\n\neffectiveness_map = {'Ineffective':0, 'Adequate':1, 'Effective':2}\ntrain['target'] = train['discourse_effectiveness'].map(effectiveness_map)\n\nsgkf = StratifiedGroupKFold(n_splits=CFG.n_folds,shuffle=True,random_state=CFG.seed)\n\nfor fold, (_,val_idx) in enumerate(sgkf.split(X=train, y=train['target'], groups=train.essay_id)):\n    train.loc[val_idx,'kfold'] = fold\n\nword2vec_model = gensim.models.KeyedVectors.load_word2vec_format('../input/google-news/GoogleNews-vectors-negative300.bin', binary=True)\nprint(word2vec_model.vectors.shape)\n\ndef avg_feature_vector(sentence, model, num_features):\n    words = sentence.replace('\\n',\" \").replace(',',' ').replace('.',\" \").split()\n    feature_vec = np.zeros((num_features,),dtype=\"float32\")\n    i=0\n    for word in words:\n        try:\n            feature_vec = np.add(feature_vec, model[word])\n        except KeyError as error:\n            feature_vec \n            i = i + 1\n    if len(words) > 0:\n        feature_vec = np.divide(feature_vec, len(words)- i)\n    return feature_vec\n\nparams = {}\nparams[\"objective\"] = 'multiclass'\nparams['metric'] = 'multi_logloss'\nparams['boosting'] = 'gbdt'\nparams['num_class'] = 3\nparams['is_unbalance'] = True\nparams[\"learning_rate\"] = 0.05\nparams[\"lambda_l2\"] = 0.0256\nparams[\"num_leaves\"] = 52\nparams[\"max_depth\"] = 10\nparams[\"feature_fraction\"] = 0.503\nparams[\"bagging_fraction\"] = 0.741\nparams[\"bagging_freq\"] = 8\nparams[\"bagging_seed\"] = 10\nparams[\"min_data_in_leaf\"] = 10\nparams[\"verbosity\"] = -1\nparams[\"random_state\"] = 42\nnum_rounds = 1000\n\noof_score = 0\ny_test_pred = np.zeros((test.shape[0], 3))\n\nfor fold in range(CFG.n_folds):\n    print(f'=============fold:{fold}==================')\n    train_fold=train[train['kfold']!=fold].reset_index(drop=True)\n    valid_fold=train[train['kfold']==fold].reset_index(drop=True)\n\n    #word2vec\n\n    #discourse_text\n    word2vec_train_disc_text = np.zeros((len(train_fold.index),300),dtype=\"float32\")\n    word2vec_valid_disc_text = np.zeros((len(valid_fold.index),300),dtype=\"float32\")\n    word2vec_test_disc_text = np.zeros((len(test.index),300),dtype=\"float32\")\n    for i in range(len(train_fold.index)):\n        word2vec_train_disc_text[i] = avg_feature_vector(train_fold[\"discourse_text\"][i], word2vec_model, 300)\n    for i in range(len(valid_fold.index)):\n        word2vec_valid_disc_text[i] = avg_feature_vector(valid_fold[\"discourse_text\"][i], word2vec_model, 300)\n    for i in range(len(test.index)):\n        word2vec_test_disc_text[i] = avg_feature_vector(test[\"discourse_text\"][i], word2vec_model, 300)\n\n    #essay_text\n    word2vec_train_essay_text = np.zeros((len(train_fold.index),300),dtype=\"float32\")\n    word2vec_valid_essay_text = np.zeros((len(valid_fold.index),300),dtype=\"float32\")\n    word2vec_test_essay_text = np.zeros((len(test.index),300),dtype=\"float32\")\n    for i in range(len(train_fold.index)):\n        word2vec_train_essay_text[i] = avg_feature_vector(train_fold[\"essay_text\"][i], word2vec_model, 300)\n    for i in range(len(valid_fold.index)):\n        word2vec_valid_essay_text[i] = avg_feature_vector(valid_fold[\"essay_text\"][i], word2vec_model, 300)\n    for i in range(len(test.index)):\n        word2vec_test_essay_text[i] = avg_feature_vector(test[\"essay_text\"][i], word2vec_model, 300)\n\n    #OneHot\n    ohe = OneHotEncoder()\n    train_type_ohe=sparse.csr_matrix(ohe.fit_transform(train_fold['discourse_type'].values.reshape(-1,1)))\n    valid_type_ohe=sparse.csr_matrix(ohe.transform(valid_fold['discourse_type'].values.reshape(-1,1)))\n    test_type_ohe=sparse.csr_matrix(ohe.transform(test['discourse_type'].values.reshape(-1,1)))\n\n\n    #merge\n    Xtrain_word2vec = sparse.hstack((train_type_ohe,word2vec_train_disc_text,word2vec_train_essay_text))\n    Xvalid_word2vec = sparse.hstack((valid_type_ohe,word2vec_valid_disc_text,word2vec_valid_essay_text))\n    test_word2vec = sparse.hstack((test_type_ohe,word2vec_test_disc_text,word2vec_test_essay_text))\n\n    #lgbm\n    lgtrain = lgb.Dataset(Xtrain_word2vec, label=train_fold['target'].ravel())\n    lgvalidation = lgb.Dataset(Xvalid_word2vec, label=valid_fold['target'].ravel())\n\n    model = lgb.train(params, lgtrain, num_rounds, \n                    valid_sets=[lgtrain, lgvalidation], \n                    early_stopping_rounds=100, verbose_eval=100)\n\n    y_pred = model.predict(Xvalid_word2vec, num_iteration=model.best_iteration)\n    y_test_pred += model.predict(test_word2vec, num_iteration=model.best_iteration)\n\n    score = log_loss(valid_fold['target'], y_pred)\n    oof_score += score\n\n    print(f'Fold:{fold},valid score:{score}')\n    \ny_test_pred = y_test_pred / float(CFG.n_folds)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:09:00.787384Z","iopub.execute_input":"2022-08-11T05:09:00.788112Z","iopub.status.idle":"2022-08-11T05:25:53.434918Z","shell.execute_reply.started":"2022-08-11T05:09:00.788071Z","shell.execute_reply":"2022-08-11T05:25:53.433748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_ineffective = y_test_pred[:,0]\nlgbm_adequate = y_test_pred[:,1]\nlgbm_effective = y_test_pred[:,2]\n\nlgbm_ineffective_non = pd.DataFrame(lgbm_ineffective)\nlgbm_adequate_non = pd.DataFrame(lgbm_adequate)\nlgbm_effective_non = pd.DataFrame(lgbm_effective)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:25:53.437277Z","iopub.execute_input":"2022-08-11T05:25:53.438340Z","iopub.status.idle":"2022-08-11T05:25:53.444870Z","shell.execute_reply.started":"2022-08-11T05:25:53.438299Z","shell.execute_reply":"2022-08-11T05:25:53.443925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"level_names = ['deberta', 'roberta', 'xg','lgbm']\n\nineffective_ = pd.concat(\n    [deb_ineffective, rob_ineffective, xg_in,lgbm_ineffective_non],\n    keys=level_names, axis=1\n)\n\nadequate_ = pd.concat(\n    [deb_adequate, rob_adequate, xg_ad,lgbm_adequate_non],\n    keys=level_names, axis=1\n)\n\neffective_ = pd.concat(\n    [deb_effective, rob_effective, xg_ef,lgbm_effective_non],\n    keys=level_names, axis=1\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:25:53.446491Z","iopub.execute_input":"2022-08-11T05:25:53.446940Z","iopub.status.idle":"2022-08-11T05:25:53.464607Z","shell.execute_reply.started":"2022-08-11T05:25:53.446878Z","shell.execute_reply":"2022-08-11T05:25:53.463639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission_origin.copy()\n\nw_ = [0.65,0.05,0.12,0.10]  # ['deberta', 'roberta', 'xg','lgbm']\nd_ = [('Ineffective', ineffective_),\n      ('Adequate', adequate_),\n      ('Effective', effective_)]\n\nfor x in d_:\n    col_name, df = x\n    submission[col_name] = pd.DataFrame({col: (df[col].mean(axis=1)) for col in level_names}).mul(w_).sum(axis=1)    \n\nsubmission.head(N_ROW)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:30:31.683399Z","iopub.execute_input":"2022-08-11T05:30:31.683767Z","iopub.status.idle":"2022-08-11T05:30:31.718203Z","shell.execute_reply.started":"2022-08-11T05:30:31.683736Z","shell.execute_reply":"2022-08-11T05:30:31.717129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:31:28.624916Z","iopub.execute_input":"2022-08-11T05:31:28.625624Z","iopub.status.idle":"2022-08-11T05:31:28.634119Z","shell.execute_reply.started":"2022-08-11T05:31:28.625586Z","shell.execute_reply":"2022-08-11T05:31:28.633046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}