{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%writefile data_processing.py\nimport pandas as pd\nimport numpy as np\nimport random\nimport gc\nimport time\nimport shutil\nimport re\nimport os\nfrom tqdm import tqdm\nimport glob\n# from unidecode import unidecode\nfrom parameter import Parameter\nfrom utils import KFold\n\nparameter = Parameter()\n\n\ndef get_data(seed=27, mode=0):\n    if os.path.exists('../input/ai4codetrainpicklefile/train_df.pkl'):\n        train_df = pd.read_pickle('../input/ai4codetrainpicklefile/train_df.pkl')\n    else:\n        train_df = read_json_data(mode='train')\n        train_orders = pd.read_csv(parameter.data_dir + 'train_orders.csv')\n        train_ancestors = pd.read_csv(parameter.data_dir + 'train_ancestors.csv')\n\n        train_orders['cell_id'] = train_orders['cell_order'].str.split()\n        train_orders = train_orders.explode(column='cell_id')\n        train_orders['rank'] = train_orders.groupby(by='id').cumcount() + 1\n        # train_orders['flag'] = range(len(train_orders))\n        # train_orders['rank'] = train_orders.groupby(by=['id'])['flag'].rank(ascending=True, method='first').astype(int)\n        # del train_orders['flag'], train_orders['cell_order']\n        del train_orders['cell_order']\n        print(train_orders)\n        # train_df = preprocess_features(train_df)\n        train_df = train_df.merge(train_orders, on=['id', 'cell_id'], how='left')\n        train_df = train_df.merge(train_ancestors[['id', 'ancestor_id']], on=['id'], how='left')\n        train_df.to_pickle('train_df.pkl')\n\n    train_df = KFold(seed, parameter.k_folds).group_split(train_df, group_col='ancestor_id')\n    # train_df = preprocess_features(train_df)\n    # train_df['source_length'] = train_df['source'].apply(len)\n    # train_df['id_length'] = train_df.groupby(by=['id'])['source_length'].transform('sum')\n    train_df = preprocess_df(train_df)\n    train_df = pd.concat(\n        [train_df[train_df['cell_type'] == 0], train_df[train_df['cell_type'] == 1].sample(frac=1.0)]).reset_index(\n        drop=True)\n    train_df['rank2'] = (train_df.groupby(by=['id', 'cell_type']).cumcount() + 1) / \\\n                        train_df.groupby(by=['id', 'cell_type'])['cell_id'].transform('count')\n    train_df.loc[train_df['cell_type'] == 1, 'rank2'] = -1\n    code_df_valid = train_df[train_df['cell_type'] == 0][['id', 'cell_id', 'rank2']].copy()\n    \n    for col in ['cell_count','markdown_count', 'code_count']:\n        train_df[col] = (train_df[col] - train_df[col].mean())/ train_df[col].std()\n        train_df[col] = np.clip(train_df[col].fillna(0.0), -3, 3)\n\n    train_df = get_truncated_df(train_df, cell_count=parameter.cell_count)\n#     train_df['flag'] = train_df['cell_type'].apply(lambda x:np.sum(x))\n#     train_df = train_df[train_df['flag']>0]\n#     del train_df['flag']\n    print(train_df)\n    print(train_df.shape)\n    return train_df, code_df_valid\n\n\ndef read_json_data(mode='train'):\n    paths_train = sorted(list(glob.glob(parameter.data_dir + '{}/*.json'.format(mode))))  # [:100]\n    res = pd.concat([\n        pd.read_json(path, dtype={'cell_type': 'category', 'source': 'str'}).assign(\n            id=path.split('/')[-1].split('.')[0]).rename_axis('cell_id')\n        for path in tqdm(paths_train)]).reset_index(drop=False)\n    res = res[['id', 'cell_id', 'cell_type', 'source']]\n    return res\n\n\ndef preprocess_df(df):\n    df['cell_count'] = df.groupby(by=['id'])['cell_id'].transform('count')\n    # df['source'] = df['cell_type'] + ' ' + df['source']\n    df['cell_type'] = df['cell_type'].map({'code': 0, 'markdown': 1}).fillna(0).astype(int)\n    # df.loc[df['cell_type']==0, 'source'] = df.loc[df['cell_type']==0, 'rank'] + ' ' + df.loc[df['cell_type']==0, 'source']\n    df['markdown_count'] = df.groupby(by=['id'])['cell_type'].transform('sum')\n    df['code_count'] = df['cell_count'] - df['markdown_count']\n    df['rank'] = df['rank'] / df['cell_count']\n    df['source'] = df['source'].apply(lambda x: x.lower().strip())\n    df['source'] = df['source'].apply(lambda x:preprocess_text(x))\n    # df['source'] = df['source'].replace(\"\\\\n\", \"\\n\")\n    # df['source'] = df['source'].str.replace(\"\\n\", \"\")\n    df['source'] = df['source'].str.replace(\"[SEP]\", \"\")\n    df['source'] = df['source'].str.replace(\"[CLS]\", \"\")\n\n    # df['source'] = df['source'].replace(\"#\", \"\")\n    # df['source'] = df['source'].apply(lambda x: unidecode(x))\n    df['source'] = df['source'].apply(lambda x: re.sub(' +', ' ', x))\n    return df\n\n# from https://www.kaggle.com/code/ilyaryabov/fastttext-sorting-with-cosine-distance-algo\nimport re\nfrom nltk.stem import WordNetLemmatizer\n\nstemmer = WordNetLemmatizer()\n\ndef preprocess_text(document):\n        # Remove all the special characters\n        document = re.sub(r'\\W', ' ', str(document))\n        document = document.replace('_',' ')\n\n        # remove all single characters\n        document = re.sub(r'\\s+[a-zA-Z]\\s+', ' ', document)\n\n#         # Remove single characters from the start\n#         document = re.sub(r'\\^[a-zA-Z]\\s+', ' ', document)\n\n        # Substituting multiple spaces with single space\n        document = re.sub(r'\\s+', ' ', document, flags=re.I)\n\n#         # Removing prefixed 'b'\n#         document = re.sub(r'^b\\s+', '', document)\n\n        # Converting to Lowercase\n        document = document.lower()\n        #return document\n\n#         # Lemmatization\n#         tokens = document.split()\n#         tokens = [stemmer.lemmatize(word) for word in tokens]\n#         # tokens = [word for word in tokens if len(word) > 3]\n\n#         preprocessed_text = ' '.join(tokens)\n        return document\n\ndef get_truncated_df(df, cell_count=128, id_col='id2', group_col='id', max_random_cnt=100, expand_ratio=5):\n    tmp1 = df[df['cell_count'] <= cell_count].reset_index(drop=True)\n    tmp1.loc[:, id_col] = 1\n    tmp2 = df[df['cell_count'] > cell_count].reset_index(drop=True)\n    # print(tmp1.shape,tmp2.shape)\n    res = [tmp1]\n    for _, df_g in tmp2.groupby(by=group_col):\n        # print(df_g.columns)\n        df_g = df_g.sample(frac=1.0).reset_index(drop=True)\n        step = min(cell_count // 2, len(df_g) - cell_count)\n        step = max(step, 1)\n        id_col_count = 1\n        for i in range(0, len(df_g), step):\n            res_tmp = df_g.iloc[i:i + cell_count]  # .copy()\n            if len(res_tmp) != cell_count:\n                res_tmp = df_g.iloc[-cell_count:]\n            # if len(res_tmp) == cell_count:\n            res_tmp.loc[:, id_col] = id_col_count\n            id_col_count += 1\n            res.append(res_tmp)\n            if i + cell_count >= len(df_g):\n                break\n\n        if len(df_g) // cell_count > 1.3:\n            random_cnt = int(len(df_g) // cell_count * expand_ratio)\n            random_cnt = min(random_cnt, max_random_cnt)  # todo\n\n            for i in range(random_cnt):\n                res_tmp = df_g.sample(n=cell_count).reset_index(drop=True)\n                res_tmp.loc[:, id_col] = id_col_count\n                id_col_count += 1\n                res.append(res_tmp)\n\n    res = pd.concat(res).reset_index(drop=True)\n    res = res.sort_values(by=['id', id_col, 'cell_type', 'rank2'], ascending=True)\n    res = res.groupby(by=['id', id_col, 'fold_flag', 'cell_count', 'markdown_count', 'code_count'], as_index=False, sort=False)[\n        ['cell_id', 'cell_type', 'source', 'rank', 'rank2']].agg(list)\n    return res\n\n\ndef get_truncated_df2(df, cell_count=128, id_col='id2', group_col='id'):\n    res = []\n    for _, df_g in df.groupby(by=group_col):\n        # print(df_g.columns)\n        df_g = df_g.reset_index(drop=True)\n        step = cell_count\n        step = max(step, 1)\n        id_col_count = 1\n        for i in range(0, len(df_g), step):\n            res_tmp = df_g.iloc[i:i + cell_count]\n            if len(res_tmp) >0:\n                res_tmp.loc[:, id_col] = id_col_count\n                id_col_count += 1\n                res.append(res_tmp)\n                if i + cell_count >= len(df_g):\n                    break\n    res = pd.concat(res).reset_index(drop=True)\n    res = res.sort_values(by=['id', id_col, 'cell_type', 'rank2'], ascending=True)\n    res = res.groupby(by=['id', id_col], as_index=False, sort=False)[['cell_id', 'cell_type', 'source']].agg(list)\n    return res","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T09:48:28.810182Z","iopub.execute_input":"2022-07-26T09:48:28.810554Z","iopub.status.idle":"2022-07-26T09:48:28.824975Z","shell.execute_reply.started":"2022-07-26T09:48:28.810522Z","shell.execute_reply":"2022-07-26T09:48:28.824106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile dataset.py\nimport pandas as pd\nimport torch\nimport random\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data.sampler import Sampler\nimport os\n# from utils import create_label\nimport numpy as np\n\n\nclass MarkdownDataset(Dataset):\n\n    def __init__(self, meta_data: pd.DataFrame, tokenizer, fold: int = -1, mode='train', parameter=None):\n        self.meta_data = meta_data.copy()\n        self.meta_data.reset_index(drop=True, inplace=True)\n        if mode == 'train':\n            self.meta_data = self.meta_data[self.meta_data['fold_flag'] != fold].copy()\n            self.meta_data = self.meta_data.iloc[:60000]\n        elif mode == 'valid':\n            self.meta_data = self.meta_data[self.meta_data['fold_flag'] == fold].copy()\n            self.meta_data = self.meta_data[self.meta_data['id'].isin(self.meta_data['id'].values[:1000])]\n        elif mode == 'test':\n            pass\n        else:\n            raise ValueError(mode)\n        self.meta_data.reset_index(drop=True, inplace=True)\n        if tokenizer.sep_token != '[SEP]':\n            self.meta_data['source'] = self.meta_data['source'].apply(\n                lambda x: [\n                    y.replace(tokenizer.sep_token, '').replace(tokenizer.cls_token, '').replace(tokenizer.pad_token, '')\n                    for y in x])\n        self.parameter = parameter\n        self.seq_length = parameter.seq_length\n        self.source = self.meta_data['source'].values\n        self.cell_type = self.meta_data['cell_type'].values\n        # self.cell_id = self.meta_data['cell_id'].values\n        self.rank = self.meta_data['rank'].values\n        # self.dense_features = self.meta_data[['cell_count','markdown_count', 'code_count']].values\n        self.mode = mode\n        self.tokenizer = tokenizer\n\n    def __getitem__(self, index):\n        source = self.source[index]\n        cell_type = self.cell_type[index]\n        rank = self.rank[index]\n        # dense_features = 1#self.dense_features[index]\n#         if self.mode == 'train':\n#             range_tmp1 = [ i for i in range(len(cell_type)) if cell_type[i]==0]\n#             range_tmp2 = [ i for i in range(len(cell_type)) if cell_type[i]==1]\n#             np.random.shuffle(range_tmp2)\n#             source = [source[i] for i in range_tmp1 + range_tmp2]\n#             rank = [rank[i] for i in range_tmp1 + range_tmp2]\n#             rank2 = [rank2[i] for i in range_tmp1 + range_tmp2]\n\n        cell_inputs = self.tokenizer.batch_encode_plus(\n            source,\n            add_special_tokens=False,\n            max_length=self.parameter.cell_max_length,\n            # padding=\"max_length\",\n            return_attention_mask=False,\n            truncation=True,\n        )\n        seq, seq_mask, target_mask, target = self.max_length_rule_base(cell_inputs['input_ids'],\n                                                                                       cell_type, rank)\n        # print(seq, seq_mask, dense_features, target_mask, target)\n        # if self.mode == 'train':\n        #     attention_mask, target = self.random_mask(attention_mask, target)\n        # print(encoded)\n        # print(target)\n        return seq, seq_mask, target_mask, target\n        # return encoded['input_ids'][0], encoded['attention_mask'][0], np.array(target, dtype=np.float32)\n\n    def __len__(self):\n        return len(self.meta_data)\n\n    def max_length_rule_base(self, cell_inputs, cell_type, rank):\n        init_length = [len(x) for x in cell_inputs]\n        total_max_length = self.seq_length - len(init_length)\n        min_length = total_max_length // len(init_length)\n        cell_length = self.search_length(init_length, min_length, total_max_length, len(init_length))\n        # print(init_code_length,code_length)\n\n        seq = []\n        for i in range(len(cell_length)):\n            if cell_type[i] == 0:\n                seq.append(self.tokenizer.cls_token_id)\n            else:\n                seq.append(self.tokenizer.sep_token_id)\n\n            if cell_length[i] > 0:\n                seq.extend(cell_inputs[i][:cell_length[i]])\n\n        # print(len(seq),'1111', np.sum(init_length),np.sum(cell_length))\n#         if len(seq) < self.seq_length:\n#             seq_mask = [1] * len(seq) + [0] * (self.seq_length - len(seq))\n#             seq = seq + [self.tokenizer.pad_token_id] * (self.seq_length - len(seq))\n#         else:\n#             seq_mask = [1] * self.seq_length\n#             seq = seq[:self.seq_length]\n        seq, seq_mask = np.array(seq, dtype=np.int), np.array(seq_mask, dtype=np.int)\n        target_mask = np.where((seq == self.tokenizer.cls_token_id) | (seq == self.tokenizer.sep_token_id), 1, 0)  # todo\n        target = np.zeros(len(seq), dtype=np.float32)\n        tmp = np.where((seq == self.tokenizer.cls_token_id) | (seq == self.tokenizer.sep_token_id))\n        target[tmp] = rank\n        sample_weight = np.zeros(len(seq), dtype=np.float32)\n        sample_weight = np.where(seq == self.tokenizer.cls_token_id, 0.33, sample_weight)\n        sample_weight = np.where(seq == self.tokenizer.sep_token_id, 1.0, sample_weight)\n#         dense_features = np.zeros(self.seq_length, dtype=np.float32)\n#         dense_features[tmp] = rank2\n        return seq, seq_mask, target_mask, target, sample_weight\n\n    @staticmethod\n    def search_length(init_length, min_length, total_max_length, cell_count, step=4, max_search_count=50):\n        if np.sum(init_length) <= total_max_length:\n            return init_length\n\n        res = [min(init_length[i], min_length) for i in range(cell_count)]\n        for s_i in range(max_search_count):\n            tmp = [min(init_length[i], res[i] + step) for i in range(cell_count)]\n            if np.sum(tmp) < total_max_length:\n                res = tmp\n            else:\n                break\n        for s_i in range(cell_count):\n            tmp = [i for i in res]\n            tmp[s_i] = min(init_length[s_i], res[s_i] + step)\n            if np.sum(tmp) < total_max_length:\n                res = tmp\n            else:\n                break\n        return res\n    \n\nclass MarkdownDatasetV2(Dataset):\n\n    def __init__(self, meta_data: pd.DataFrame, tokenizer, parameter=None, max_length=4096):\n        self.meta_data = meta_data.copy()\n        self.meta_data.reset_index(drop=True, inplace=True)\n        if tokenizer.sep_token != '[SEP]':\n            self.meta_data['source'] = self.meta_data['source'].apply(\n                lambda x: [\n                    y.replace(tokenizer.sep_token, '').replace(tokenizer.cls_token, '').replace(tokenizer.pad_token, '')\n                    for y in x])\n        self.batch_max_length = self.meta_data['batch_max_length'].values\n        self.source = self.meta_data['source'].values\n        self.parameter = parameter\n        self.max_length = max_length\n        self.cell_type = self.meta_data['cell_type'].values\n        # self.cell_id = self.meta_data['cell_id'].values\n        self.tokenizer = tokenizer\n\n    def __getitem__(self, index):\n        source = self.source[index]\n        cell_type = self.cell_type[index]\n        batch_max_len = min(self.batch_max_length[index], self.max_length)\n\n        cell_inputs = self.tokenizer.batch_encode_plus(\n            source,\n            add_special_tokens=False,\n            max_length=self.parameter.cell_max_length,\n            # padding=\"max_length\",\n            return_attention_mask=False,\n            truncation=True,\n        )\n        seq, seq_mask, target_mask = self.max_length_rule_base(cell_inputs['input_ids'], cell_type, batch_max_len)\n        return seq, seq_mask, target_mask\n\n    def __len__(self):\n        return len(self.meta_data)\n\n    def max_length_rule_base(self, cell_inputs, cell_type, batch_max_len):\n        init_length = [len(x) for x in cell_inputs]\n        total_max_length = batch_max_len - len(init_length)\n        min_length = total_max_length // len(init_length)\n        cell_length = self.search_length(init_length, min_length, total_max_length, len(init_length))\n        # print(init_code_length,code_length)\n\n        seq = []\n        for i in range(len(cell_length)):\n            if cell_type[i] == 0:\n                seq.append(self.tokenizer.cls_token_id)\n            else:\n                seq.append(self.tokenizer.sep_token_id)\n\n            if cell_length[i] > 0:\n                seq.extend(cell_inputs[i][:cell_length[i]])\n\n        # print(len(seq),'1111', np.sum(init_length),np.sum(cell_length))\n        if len(seq) < batch_max_len:\n            seq_mask = [1] * len(seq) + [0] * (batch_max_len - len(seq))\n            seq = seq + [self.tokenizer.pad_token_id] * (batch_max_len - len(seq))\n        else:\n            seq_mask = [1] * batch_max_len\n            seq = seq[:batch_max_len]\n        seq, seq_mask = np.array(seq, dtype=np.int), np.array(seq_mask, dtype=np.int)\n        target_mask = np.where((seq == self.tokenizer.cls_token_id) | (seq == self.tokenizer.sep_token_id), 1, 0)\n        return seq, seq_mask, target_mask\n\n    @staticmethod\n    def search_length(init_length, min_length, total_max_length, cell_count, step=4, max_search_count=50):\n        if np.sum(init_length) <= total_max_length:\n            return init_length\n\n        res = [min(init_length[i], min_length) for i in range(cell_count)]\n        for s_i in range(max_search_count):\n            tmp = [min(init_length[i], res[i] + step) for i in range(cell_count)]\n            if np.sum(tmp) < total_max_length:\n                res = tmp\n            else:\n                break\n        for s_i in range(cell_count):\n            tmp = [i for i in res]\n            tmp[s_i] = min(init_length[s_i], res[s_i] + step)\n            if np.sum(tmp) < total_max_length:\n                res = tmp\n            else:\n                break\n        return res","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:28.995115Z","iopub.execute_input":"2022-07-26T09:48:28.995615Z","iopub.status.idle":"2022-07-26T09:48:29.008983Z","shell.execute_reply.started":"2022-07-26T09:48:28.995578Z","shell.execute_reply":"2022-07-26T09:48:29.008105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile models.py\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport math\nfrom transformers import RobertaModel, RobertaConfig, AutoConfig, AutoModel, AutoModelForMaskedLM\n\n\nclass MarkdownModel(nn.Module):\n    def __init__(self, name, num_classes=1, seq_length=96, pretrained=True):\n        super(MarkdownModel, self).__init__()\n        # self.encoder = AutoModel.from_pretrained(name, attention_probs_dropout_prob=0.1, hidden_dropout_prob=0.1)\n        self.config = AutoConfig.from_pretrained(name)\n        self.config.attention_probs_dropout_prob = 0.\n        self.config.hidden_dropout_prob = 0.\n        self.config.max_position_embeddings = 4096 * 2 \n        # self.config.output_hidden_states = True\n        if pretrained:\n            self.encoder = AutoModel.from_pretrained(name, config=self.config, ignore_mismatched_sizes=True)\n            # self.encoder = AutoModelForMaskedLM.from_pretrained(name, config=self.config)\n        else:\n            # self.encoder = AutoModelForMaskedLM.from_config(self.config)\n            self.encoder = AutoModel.from_config(self.config)\n\n        # self.encoder = AutoModel.from_pretrained(name)\n        # print(self.encoder.__dict__)\n        # transformer_layers = 2\n#         self.seq_length = seq_length\n#         self.transformer_layers = transformer_layers\n        self.in_dim = self.encoder.config.hidden_size\n        print(self.in_dim)\n#         self.pe = PositionalEncoding(self.in_dim)\n#         self.trans = nn.Sequential(\n#             *[TransformerBlock(emb_s=64, head_cnt=self.in_dim // 64, dp1=0., dp2=0.) for _ in\n#               range(transformer_layers)])\n        self.bilstm = nn.LSTM(self.in_dim, self.in_dim, num_layers=1, \n                              dropout=self.config.hidden_dropout_prob, batch_first=True,\n                              bidirectional=True)\n        # self.dropouts = nn.ModuleList([nn.Dropout(0.5) for _ in range(5)])\n#         hidden = 64\n#         dropout = 0.\n#         self.sequence = nn.Sequential(\n#             # nn.BatchNorm1d(1),\n#             nn.Linear(1, hidden),  # todo\n#             nn.Dropout(dropout),\n#             nn.ReLU(),\n#             # nn.BatchNorm1d(hidden),\n#             nn.Linear(hidden, hidden),\n#             nn.Dropout(dropout),\n#             nn.ReLU()\n#         )\n        self.last_fc = nn.Linear(self.in_dim*2, num_classes)\n        # self.fc = nn.LazyLinear(num_classes)\n        torch.nn.init.normal_(self.last_fc.weight, std=0.02)\n        self.sig = nn.Sigmoid()\n\n    def forward(self, x, mask):\n        x = self.encoder(x, attention_mask=mask)[\"last_hidden_state\"]\n        # x = x.reshape(-1, code_count, self.seq_length, self.in_dim).mean(2)\n        #         x = torch.sum(x * mask.unsqueeze(-1), dim=1) / torch.sum(mask, dim=1).unsqueeze(-1)\n        #         x = x.reshape(-1, code_count, self.in_dim)\n        # x = x + self.sequence(dense_features.unsqueeze(-1))\n        # print(x)\n        # print(x.shape)\n#         prev = None\n#         x = self.pe(x)\n#         for i in range(self.transformer_layers):\n#             # x = x * mask.unsqueeze(-1)\n#             x, prev = self.trans[i](x, prev)\n        # x = torch.sum(x * mask.unsqueeze(-1), dim=1) / torch.sum(mask, dim=1).unsqueeze(-1)\n        # x = torch.cat([x, self.sequence(dense_features.unsqueeze(1)).repeat(1,2048,1)], dim=2)\n        # x = x.mean(1)\n        x, _ = self.bilstm(x)\n        out = self.last_fc(x)\n#         for i, dropout in enumerate(self.dropouts):\n#             if i == 0:\n#                 out = self.last_fc(dropout(x))\n#             else:\n#                 out += self.last_fc(dropout(x))\n#         out /= len(self.dropouts)\n        # out = self.sig(out)\n        out = out.squeeze(-1)\n        return out\n# input = torch.randn(2, 200).long() +10\n# input2 = torch.zeros(2, 200)\n# net = MarkdownModel('roberta-base', pretrained=False)\n# print(input)\n# print(net(input, input2))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:29.012097Z","iopub.execute_input":"2022-07-26T09:48:29.012377Z","iopub.status.idle":"2022-07-26T09:48:29.028968Z","shell.execute_reply.started":"2022-07-26T09:48:29.012352Z","shell.execute_reply":"2022-07-26T09:48:29.028188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile parameter.py\nimport torch\n\n\nclass Parameter(object):\n    def __init__(self):\n        # data\n        self.result_dir = './user_data/'\n        self.data_dir = '../input/AI4Code/'\n        self.k_folds = 5\n        self.n_jobs = 4\n        self.random_seed = 27\n        self.seq_length = 512\n        self.cell_count = 128\n        self.cell_max_length = 128\n        self.device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\n        # model\n        self.use_cuda = torch.cuda.is_available()\n        self.gpu = 0\n        self.print_freq = 100\n        self.lr = 0.003\n        self.weight_decay = 0\n        self.optim = 'Adam'\n        self.base_epoch = 30\n\n    def get(self, name):\n        return getattr(self, name)\n\n    def set(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def __str__(self):\n        return '\\n'.join(['%s:%s' % item for item in self.__dict__.items()])\n\n\nif __name__ == '__main__':\n    parameter = Parameter()\n    print(parameter)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:29.052857Z","iopub.execute_input":"2022-07-26T09:48:29.053215Z","iopub.status.idle":"2022-07-26T09:48:29.05951Z","shell.execute_reply.started":"2022-07-26T09:48:29.053187Z","shell.execute_reply":"2022-07-26T09:48:29.058686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile utils.py\nimport pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, OneHotEncoder\nimport sys\nimport os\nimport random\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim.lr_scheduler import LambdaLR\nimport itertools\n\n\nclass KFold(object):\n    \"\"\"\n    KFold: Group split by group_col or random_split\n    \"\"\"\n\n    def __init__(self, random_seed, k_folds=10, flag_name='fold_flag'):\n        self.k_folds = k_folds\n        self.flag_name = flag_name\n        np.random.seed(random_seed)\n\n    def group_split(self, train_df, group_col):\n        group_value = list(set(train_df[group_col]))\n        group_value.sort()\n        fold_flag = [i % self.k_folds for i in range(len(group_value))]\n        np.random.shuffle(fold_flag)\n        train_df = train_df.merge(pd.DataFrame({group_col: group_value, self.flag_name: fold_flag}), how='left',\n                                  on=group_col)\n        return train_df\n\n    def random_split(self, train_df):\n        fold_flag = [i % self.k_folds for i in range(len(train_df))]\n        np.random.shuffle(fold_flag)\n        train_df[self.flag_name] = fold_flag\n        return train_df\n\n    def stratified_split(self, train_df, group_col):\n        train_df[self.flag_name] = 1\n        train_df[self.flag_name] = train_df.groupby(by=[group_col])[self.flag_name].rank(ascending=True,\n                                                                                         method='first').astype(int)\n        train_df[self.flag_name] = train_df[self.flag_name].sample(frac=1.0).reset_index(drop=True)\n        train_df[self.flag_name] = train_df[self.flag_name] % self.k_folds\n        return train_df\n\n\n# http://stackoverflow.com/questions/34950201/pycharm-print-end-r-statement-not-working\nclass Logger(object):\n    def __init__(self):\n        self.terminal = sys.stdout  # stdout\n        self.file = None\n\n    def open(self, file, mode=None):\n        if mode is None: mode = 'w'\n        self.file = open(file, mode)\n\n    def write(self, message, is_terminal=1, is_file=1):\n        if '\\r' in message: is_file = 0\n\n        if is_terminal == 1:\n            self.terminal.write(message)\n            self.terminal.flush()\n            # time.sleep(1)\n\n        if is_file == 1:\n            self.file.write(message)\n            self.file.flush()\n\n    def flush(self):\n        # this flush method is needed for python 3 compatibility.\n        # this handles the flush command by doing nothing.\n        # you might want to specify some extra behavior here.\n        pass\n\n\ndef seed_everything(random_seed):\n    random.seed(random_seed)\n    np.random.seed(random_seed)\n    torch.manual_seed(random_seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(random_seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(random_seed)\n        torch.cuda.manual_seed_all(random_seed)\n        #         torch.backends.cudnn.enabled = False\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n\n\nclass AverageMeter(object):\n    \"\"\"Computes and stores the average and current value\"\"\"\n\n    def __init__(self):\n        self.reset()\n\n    def reset(self):\n        self.val = 0.\n        self.avg = 0.\n        self.sum = 0.\n        self.count = 0\n\n    def update(self, val, n=1):\n        self.val = val\n        self.sum += val * n\n        self.count += n\n        self.avg = self.sum / self.count\n\n\ndef save_model(model, save_path, model_name):\n    if not os.path.exists(save_path):\n        os.makedirs(save_path)\n    filename = os.path.join(save_path, model_name + '.pth.tar')\n    torch.save({'state_dict': model.state_dict(), }, filename)\n    # if is_best:\n    #     best_filename = os.path.join(save_path, model_name + '_best_model.pth.tar')\n    #     shutil.copyfile(filename, best_filename)\n\n\ndef load_model(model, load_path, model_name):\n    if not os.path.exists(load_path):\n        os.makedirs(load_path)\n    filename = os.path.join(load_path, model_name + '.pth.tar')\n    model.load_state_dict(torch.load(filename)['state_dict'])\n    return model\n\n\ndef adjust_learning_rate(optimizer, epoch, args):\n    \"\"\"Sets the learning rate to the initial LR decayed every 10 epochs\"\"\"\n    # lr = args.lr * (0.5 ** (epoch // 10))\n    for param_group in optimizer.param_groups:\n        param_group['lr'] = param_group['lr'] * (0.3 ** (epoch // 10))\n\n\ndef worker_init_fn(worker_id):\n    \"\"\"\n    Handles PyTorch x Numpy seeding issues.\n\n    Args:\n        worker_id (int): Id of the worker.\n    \"\"\"\n    np.random.seed(np.random.get_state()[1][0] + worker_id)\n\n\nclass MyLoss(nn.Module):\n    def __init__(self):\n        super().__init__()\n\n    def forward(self, inputs, rank, rank_mask):\n        # loss = (inputs - rank) ** 2 * rank_mask\n        loss = torch.abs(inputs - rank) * rank_mask\n        loss = torch.sum(loss, dim=1) / (torch.sum(rank_mask, dim=1) + 1)\n        loss = loss.mean()\n        return loss\n\n\nclass MyBCELoss(nn.Module):\n    def __init__(self, class_weight=False):\n        super().__init__()\n        self.class_weight = class_weight\n\n    def forward(self, inputs, targets, mask, sample_weight=None):\n        # print(inputs)\n        # inputs = inputs[:,:targets.shape[1]]\n        bce1 = F.binary_cross_entropy(inputs, torch.ones_like(inputs), reduction='none')\n        bce2 = F.binary_cross_entropy(inputs, torch.zeros_like(inputs), reduction='none')\n        bce = 1 * bce1 * targets + bce2 * (1 - targets)\n        # mask = torch.where(targets >= 0, torch.ones_like(bce), torch.zeros_like(bce))\n        bce = bce * mask\n        # print(bce)\n        #         if sample_weight is not None:\n        #             bce = bce * sample_weight.unsqueeze(1)\n        loss = bce.mean()  # .sum() / mask.sum()\n        return loss\n\n\nclass FGM():\n    def __init__(self, model):\n        self.model = model\n        self.backup = {}\n\n    def attack(self, epsilon=1., emb_name='emb'):\n        # emb_name这个参数要换成你模型中embedding的参数名\n        for name, param in self.model.named_parameters():\n            if param.requires_grad and emb_name in name and param.grad is not None:\n                # print(name, param)\n                self.backup[name] = param.data.clone()\n                norm = torch.norm(param.grad)\n                if norm != 0 and not torch.isnan(norm):\n                    r_at = epsilon * param.grad / max(norm, 0.001)\n                    param.data.add_(r_at)\n\n    def restore(self, emb_name='emb'):\n        # emb_name这个参数要换成你模型中embedding的参数名\n        for name, param in self.model.named_parameters():\n            if param.requires_grad and emb_name in name and param.grad is not None:\n                assert name in self.backup\n                param.data = self.backup[name]\n        self.backup = {}\n\n\nfrom bisect import bisect\n\n\n# from https://www.kaggle.com/code/ryanholbrook/competition-metric-kendall-tau-correlation\n# Actually O(N^2), but fast in practice for our data\ndef count_inversions(a):\n    inversions = 0\n    sorted_so_far = []\n    for i, u in enumerate(a):  # O(N)\n        j = bisect(sorted_so_far, u)  # O(log N)\n        inversions += i - j\n        sorted_so_far.insert(j, u)  # O(N)\n    return inversions\n\n\ndef kendall_tau(ground_truth, predictions):\n    total_inversions = 0  # total inversions in predicted ranks across all instances\n    total_2max = 0  # maximum possible inversions across all instances\n    for gt, pred in zip(ground_truth, predictions):\n        assert len(gt) == len(pred)\n        ranks = [gt.index(x) for x in pred]  # rank predicted order in terms of ground truth\n        total_inversions += count_inversions(ranks)\n        n = len(gt)\n        total_2max += n * (n - 1)\n    return 1 - 4 * total_inversions / total_2max\n\n\ndef get_score(df, masks, rank_pred, code_df_valid):\n    df['cell_id2'] = [[y[i] for i in range(len(x)) if x[i] == 1] for x, y in\n                          zip(df['cell_type'].values, df['cell_id'].values)]\n    df = df[['id', 'cell_id2']].explode('cell_id2')\n    df = df[~pd.isnull(df['cell_id2'])]\n    preds = rank_pred.flatten()[np.where(masks.flatten() == 1)]\n    df['rank2'] = preds\n    df = df.groupby(by=['id', 'cell_id2'], as_index=False)['rank2'].agg('mean')\n\n    df.rename(columns={'cell_id2': 'cell_id'}, inplace=True)\n    code_df_valid_tmp = code_df_valid[code_df_valid['id'].isin(df['id'])]\n    df = pd.concat([df[['id', 'cell_id', 'rank2']], code_df_valid_tmp]).reset_index(drop=True)\n    df = df.sort_values(by=['id', 'rank2'], ascending=True)\n    res = df.groupby(by=['id'], sort=False, as_index=False)['cell_id'].agg(list)\n\n    train_orders = pd.read_csv('../input/AI4Code/train_orders.csv')\n    train_orders['cell_order'] = train_orders['cell_order'].str.split()\n    res = res.merge(train_orders, how='left', on='id')\n    print(res)\n    score = kendall_tau(res['cell_order'], res['cell_id'])\n    return score\n\n\n# # https://www.kaggle.com/code/anyai28/fast-inference-by-padding-optimization\n# def get_sorted_test_df(df, extract_col, tokenizer, batch_size):\n#     input_lengths = []\n#     for text in df[extract_col].fillna(\"\").values:\n#         length = len(tokenizer(text, add_special_tokens=True)['input_ids'])\n#         input_lengths.append(length)\n#     df['input_lengths'] = input_lengths\n#     length_sorted_idx = np.argsort([-l for l in input_lengths])\n\n#     # sort dataframe\n#     sort_df = df.iloc[length_sorted_idx]\n#     # calc max_len per batch\n#     sorted_input_length = sort_df['input_lengths'].values\n#     batch_max_length = np.zeros_like(sorted_input_length)\n#     for i in range((len(sorted_input_length) // batch_size) + 1):\n#         batch_max_length[i * batch_size:(i + 1) * batch_size] = np.max(\n#             sorted_input_length[i * batch_size:(i + 1) * batch_size])\n#     sort_df['batch_max_length'] = batch_max_length\n#     return sort_df, length_sorted_idx\n\n# https://www.kaggle.com/code/anyai28/fast-inference-by-padding-optimization\ndef get_sorted_test_df(df, extract_col, tokenizer, batch_size, cell_max_length=128):\n    input_lengths = []\n    for text in df[extract_col].values:\n        # print(text)\n        tmp = tokenizer.batch_encode_plus(\n            text,\n            add_special_tokens=False,\n            max_length=cell_max_length,\n            return_attention_mask=False,\n            truncation=True,\n        )\n        init_length = [len(x) for x in tmp['input_ids']]\n        total_length = np.sum(init_length) + len(init_length)\n        input_lengths.append(total_length)\n    # print(input_lengths)\n    df['input_lengths'] = input_lengths\n    length_sorted_idx = np.argsort([-l for l in input_lengths])\n\n    # sort dataframe\n    sort_df = df.iloc[length_sorted_idx]\n    # calc max_len per batch\n    sorted_input_length = sort_df['input_lengths'].values\n    batch_max_length = np.zeros_like(sorted_input_length)\n    total_iter = len(sorted_input_length) // batch_size if len(sorted_input_length) % batch_size == 0 else (len(sorted_input_length) // batch_size) + 1\n    for i in range(total_iter):\n        batch_max_length[i * batch_size:(i + 1) * batch_size] = np.max(\n            sorted_input_length[i * batch_size:(i + 1) * batch_size])\n    sort_df['batch_max_length'] = batch_max_length\n    return sort_df, length_sorted_idx\n\n\ndef get_model_path(model_name):\n    res = '../input/'\n    if model_name in ['distilroberta-base', 'roberta-base', 'roberta-large']:\n        res += 'roberta-transformers-pytorch/' + model_name\n    elif model_name in ['bart-base', 'bart-large']:\n        res += 'bartbase' if model_name == 'bart-base' else 'bartlarge'\n        res += '/'\n    elif model_name in ['deberta-base', 'deberta-large', 'deberta-v2-xlarge', 'deberta-v2-xxlarge']:\n        res += 'deberta/' + model_name.replace('deberta-', '')\n    elif model_name in ['deberta-v3-large']:\n        res += 'deberta-v3-large/' + model_name\n    elif model_name in ['electra-base', 'electra-large']:\n        res += 'electra/' + model_name + '-discriminator'\n    elif 'albert' in model_name:\n        res += 'pretrained-albert-pytorch/' + model_name\n    elif model_name == 'funnel-large':\n        res += 'funnel-large/'\n    elif model_name == 'xlnet-base':\n        res += 'xlnet-pretrained/xlnet-pretrained/'\n    elif model_name == 'deberta-base-mnli':\n        res += 'huggingface-deberta-variants/deberta-base-mnli/deberta-base-mnli/'\n    elif model_name == 'deberta-xlarge':\n        res += 'huggingface-deberta-variants/deberta-xlarge/deberta-xlarge/'\n    elif model_name == 'codebert-base':\n        res += 'codebert-base/codebert-base/'\n    elif model_name == 'CodeBERTa-small-v1':\n        res += 'huggingface-code-models/CodeBERTa-small-v1/'\n    else:\n        raise ValueError(model_name)\n    return res","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:29.181811Z","iopub.execute_input":"2022-07-26T09:48:29.182269Z","iopub.status.idle":"2022-07-26T09:48:29.199317Z","shell.execute_reply.started":"2022-07-26T09:48:29.18224Z","shell.execute_reply":"2022-07-26T09:48:29.198397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile predict.py\n# coding=utf-8\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nimport sys\nimport gc\nimport time\nfrom transformers import BertTokenizer, RobertaTokenizerFast, AutoTokenizer\nimport torch\nfrom torch.utils.data import DataLoader\nfrom parameter import Parameter\nfrom models import MarkdownModel\nfrom dataset import MarkdownDataset, MarkdownDatasetV2\nfrom data_processing import read_json_data, preprocess_df, get_truncated_df, get_truncated_df2\nfrom utils import *\n\nparameter = Parameter()\nparameter.set(**{'batch_size': 2, 'n_jobs': 2})\nseed_everything(parameter.random_seed)\nos.environ['TOKENIZERS_PARALLELISM'] = 'false'\n\nlog_dir = './inference_log.txt'\nif os.path.exists(log_dir):\n    os.remove(log_dir)\nlog = Logger()\nlog.open(log_dir, mode='a')\n\n\ndef predict(model, data_loader, max_length):\n    # switch to evaluate mode\n    model.eval()\n    y_pred = []\n    mask = []\n    for i, batch_data in enumerate(data_loader):\n        if parameter.use_cuda:\n            batch_data = (t.cuda() for t in batch_data)\n        seq, seq_mask, target_mask = batch_data\n        outputs = model(seq, seq_mask).detach().cpu().numpy()\n        target_mask = target_mask.detach().cpu().numpy().reshape((outputs.shape[0], -1))\n        tmp1 = np.zeros((outputs.shape[0], max_length))\n        tmp1[:, :outputs.shape[1]] = outputs\n        tmp2 = np.zeros((outputs.shape[0], max_length))\n        tmp2[:, :outputs.shape[1]] = target_mask\n        \n        y_pred.append(tmp1)\n        mask.append(tmp2)\n        \n    y_pred = np.concatenate(y_pred)\n    mask = np.concatenate(mask)\n    return y_pred, mask\n\n\ndef get_preds(my_df, my_loader, my_model, model_path, max_length=4096):\n    if my_df.shape[0] > 0:\n        my_model.load_state_dict(torch.load(model_path)['state_dict'])\n    if parameter.use_cuda:\n        my_model = my_model.cuda()\n    with torch.no_grad():\n        y_pred, mask = predict(my_model, my_loader, max_length)\n    return y_pred, mask\n\ndef get_results(df, masks, rank_pred, code_df_valid):\n    df['cell_id2'] = df['cell_id']\n    df = df[['id', 'cell_id2']].explode('cell_id2')\n    df = df[~pd.isnull(df['cell_id2'])]\n    preds = rank_pred.flatten()[np.where(masks.flatten() == 1)]\n    df['rank2'] = preds\n    df = df.groupby(by=['id', 'cell_id2'], as_index=False)['rank2'].agg('mean')\n\n    df.rename(columns={'cell_id2': 'cell_id'}, inplace=True)\n    code_df_valid_tmp = code_df_valid[code_df_valid['id'].isin(df['id'])]\n    code_df_valid_tmp['rank3'] = code_df_valid_tmp.groupby(by=['id'])['rank2'].rank(ascending=True,method='first')\n    tmp = code_df_valid_tmp[['id','cell_id','rank3']].merge(df, how='inner',on=['id', 'cell_id'])\n    tmp['rank4'] = tmp.groupby(by=['id'])['rank2'].rank(ascending=True, method='first')\n    tmp = tmp[['id','cell_id','rank3']].merge(tmp[['id', 'rank4', 'rank2']].rename(columns={'rank4':'rank3'}), how='inner', on=['id', 'rank3'])\n    tmp = tmp[['id', 'cell_id', 'rank2']]\n    \n    df = df.merge(tmp[['id', 'cell_id','rank2']].rename(columns={'rank2':'rank3'}), how='left', on=['id', 'cell_id'])\n    df['rank2'] = np.where(pd.isnull(df['rank3']),df['rank2'], df['rank3'])\n    \n    # df = pd.concat([df[['id', 'cell_id', 'rank2']], code_df_valid_tmp]).reset_index(drop=True)\n    df = df.sort_values(by=['id', 'rank2'], ascending=True)\n    return df\n\ndef get_results2(df, masks, rank_pred, code_df_valid):\n    df = df[['id', 'id2', 'cell_id']].explode('cell_id')\n    df = df[~pd.isnull(df['cell_id'])]\n    preds = rank_pred.flatten()[np.where(masks.flatten() == 1)]\n    df['rank2'] = preds\n\n#     code_df_valid_tmp = code_df_valid[code_df_valid['id'].isin(df['id'])]\n#     code_df_valid_tmp['rank3'] = code_df_valid_tmp.groupby(by=['id'])['rank2'].rank(ascending=True,method='first')\n#     tmp = code_df_valid_tmp[['id','cell_id','rank3']].merge(df, how='inner',on=['id', 'cell_id'])\n#     tmp['rank4'] = tmp.groupby(by=['id'])['rank2'].rank(ascending=True, method='first')\n#     tmp = tmp[['id','cell_id','rank3']].merge(tmp[['id', 'rank4', 'rank2']].rename(columns={'rank4':'rank3'}), how='inner', on=['id', 'rank3'])\n#     tmp = tmp[['id', 'cell_id', 'rank2']]\n    \n#     df = df.merge(tmp[['id', 'cell_id','rank2']].rename(columns={'rank2':'rank3'}), how='left', on=['id', 'cell_id'])\n#     df['rank2'] = np.where(pd.isnull(df['rank3']),df['rank2'], df['rank3'])\n    \n    # df = pd.concat([df[['id', 'cell_id', 'rank2']], code_df_valid_tmp]).reset_index(drop=True)\n    df = df.sort_values(by=['id','id2','rank2'], ascending=True)\n    return df\n\nlog.write('>> reading test_df\\n')\ntest_df = read_json_data(mode='test')\ntest_df['rank'], test_df['fold_flag'] = 1,-1\ntest_df = preprocess_df(test_df)\n\ntest_df = pd.concat(\n        [test_df[test_df['cell_type'] == 0], test_df[test_df['cell_type'] == 1].sample(frac=1.0)]).reset_index(\n        drop=True)\ntest_df['rank2'] = (test_df.groupby(by=['id', 'cell_type']).cumcount() + 1) / \\\n                    test_df.groupby(by=['id', 'cell_type'])['cell_id'].transform('count')\ntest_df.loc[test_df['cell_type'] == 1, 'rank2'] = -1\ncode_df_sub = test_df[test_df['cell_type'] == 0][['id', 'cell_id', 'rank2']].copy()\n\n# test_df2 = test_df[test_df['cell_count']>=96]\ntest_df = get_truncated_df(test_df, cell_count=parameter.cell_count)\n\n\nlog.write('>> predicting...\\n')\nstart = time.time()\n# --------------------\nmodel_name = 'deberta-v3-large'\ntokenizer_path = get_model_path(model_name)\ntokenizer = AutoTokenizer.from_pretrained(tokenizer_path)\nsort_df, length_sorted_idx = get_sorted_test_df(test_df, 'source', tokenizer, batch_size=parameter.batch_size, cell_max_length=parameter.cell_max_length)\ndel test_df\ngc.collect()\n\n# # -------------------- part1\n# test_dataset = MarkdownDatasetV2(sort_df, tokenizer, parameter=parameter, max_length=4096)\n# test_loader = DataLoader(test_dataset, shuffle=False, batch_size=2,\n#                          num_workers=parameter.n_jobs, drop_last=False, pin_memory=True)\n# model = MarkdownModel(get_model_path(model_name), pretrained=False)\n# model_path = '../input/ai4code-model/deberta-v3-large_fold0.pth.tar'\n# y_preds, masks = get_preds(sort_df, test_loader, model, model_path, max_length=4096)\n# del model\n# gc.collect()\n# torch.cuda.empty_cache()\n\nsort_df1 = sort_df[sort_df['batch_max_length'] <= 4096]\nsort_df2 = sort_df[sort_df['batch_max_length'] > 4096]\n\n# -------------------- part1\ntest_dataset = MarkdownDatasetV2(sort_df1, tokenizer, parameter=parameter, max_length=4096)\ntest_loader = DataLoader(test_dataset, shuffle=False, batch_size=2,\n                         num_workers=parameter.n_jobs, drop_last=False, pin_memory=True)\nmodel = MarkdownModel(get_model_path(model_name), pretrained=False)\nmodel_path = '../input/ai4code-model/deberta-v3-large_fold0.pth.tar'\ny_preds, masks = get_preds(sort_df1, test_loader, model, model_path, max_length=4096+1024)\ndel model\ngc.collect()\ntorch.cuda.empty_cache()\n    \nif len(sort_df2) > 0:\n    # -------------------- part2\n    test_dataset = MarkdownDatasetV2(sort_df2, tokenizer, parameter=parameter, max_length=4096+1024)\n    test_loader = DataLoader(test_dataset, shuffle=False, batch_size=1,\n                             num_workers=parameter.n_jobs, drop_last=False, pin_memory=True)\n    model = MarkdownModel(get_model_path(model_name), pretrained=False)\n    model_path = '../input/ai4code-model/deberta-v3-large_fold0.pth.tar'\n    y_preds2, masks2 = get_preds(sort_df2, test_loader, model, model_path ,max_length=4096+1024)\n    del model\n    gc.collect()\n    torch.cuda.empty_cache()\n    y_preds = np.concatenate([y_preds2, y_preds])\n    masks = np.concatenate([masks2, masks])\n\n\nres = get_results(sort_df, masks, y_preds, code_df_sub)\n# res = pd.concat([res1, res2]).reset_index(drop=True)\n# sub_df = res1\n# sub_df = res1.sort_values(by=['id', 'rank2'], ascending=True)\nsub_df = res.groupby(by=['id'], sort=False)['cell_id'].apply(lambda x: ' '.join(x)).reset_index()\nsub_df.rename(columns={'cell_id': 'cell_order'}, inplace=True)\n# sub_df['cell_order'] = sub_df['cell_order'].apply(lambda x: ' '.join(x.split()[::-1]))\n# print(test_df.shape)\nsub_df[['id', 'cell_order']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:29.342777Z","iopub.execute_input":"2022-07-26T09:48:29.3432Z","iopub.status.idle":"2022-07-26T09:48:29.357781Z","shell.execute_reply.started":"2022-07-26T09:48:29.343169Z","shell.execute_reply":"2022-07-26T09:48:29.356537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python predict.py","metadata":{"execution":{"iopub.status.busy":"2022-07-26T09:48:29.515211Z","iopub.execute_input":"2022-07-26T09:48:29.51563Z","iopub.status.idle":"2022-07-26T09:48:52.928015Z","shell.execute_reply.started":"2022-07-26T09:48:29.515599Z","shell.execute_reply":"2022-07-26T09:48:52.926885Z"},"trusted":true},"execution_count":null,"outputs":[]}]}