{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob\nimport os\nfrom typing import List\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport transformers\nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T14:25:56.513719Z","iopub.execute_input":"2022-07-17T14:25:56.514728Z","iopub.status.idle":"2022-07-17T14:25:56.520832Z","shell.execute_reply.started":"2022-07-17T14:25:56.514677Z","shell.execute_reply":"2022-07-17T14:25:56.519831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 32\nSLICES = 8\nMD_MAX_LEN = 64\nTOTAL_MAX_LEN = 512\nSTRATEGY = tf.distribute.get_strategy()\nBASE_MODEL = \"../input/codebert-base/codebert-base\"\nTOKENIZER = transformers.AutoTokenizer.from_pretrained(BASE_MODEL)\nINPUT_PATH = \"../input/AI4Code\"","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:25:56.563377Z","iopub.execute_input":"2022-07-17T14:25:56.563676Z","iopub.status.idle":"2022-07-17T14:25:56.728547Z","shell.execute_reply.started":"2022-07-17T14:25:56.563627Z","shell.execute_reply":"2022-07-17T14:25:56.727536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_notebook(path: str) -> pd.DataFrame:\n    return (\n        pd.read_json(path, dtype={\"cell_type\": \"category\", \"source\": \"str\"})\n        .assign(id=os.path.basename(path).split(\".\")[0])\n        .rename_axis(\"cell_id\")\n    )\n\n\ndef clean_code(cell: str) -> str:\n    return str(cell).replace(\"\\\\n\", \"\\n\")\n\n\ndef sample_cells(cells: List[str], n: int) -> List[str]:\n    cells = [clean_code(cell) for cell in cells]\n    if n >= len(cells):\n        return cells\n    else:\n        results = []\n        step = len(cells) / n\n        idx = 0\n        while int(np.round(idx)) < len(cells):\n            results.append(cells[int(np.round(idx))])\n            idx += step\n        if cells[-1] not in results:\n            results[-1] = cells[-1]\n        return results\n\n\ndef get_features(df: pd.DataFrame) -> dict:\n    features = {}\n    for i, sub_df in tqdm(df.groupby(\"id\"), desc=\"Features\"):\n        features[i] = {}\n        total_md = sub_df[sub_df.cell_type == \"markdown\"].shape[0]\n        code_sub_df = sub_df[sub_df.cell_type == \"code\"]\n        total_code = code_sub_df.shape[0]\n        codes = sample_cells(code_sub_df.source.values, 20)\n        features[i][\"total_code\"] = total_code\n        features[i][\"total_md\"] = total_md\n        features[i][\"codes\"] = codes\n    return features\n\n\ndef tokenize(df: pd.DataFrame, fts: dict) -> dict:\n    input_ids = np.zeros((len(df), TOTAL_MAX_LEN), dtype=np.int32)\n    attention_mask = np.zeros((len(df), TOTAL_MAX_LEN), dtype=np.int32)\n    features = np.zeros((len(df),), dtype=np.float32)\n    \n    for i, row in tqdm(\n            df.reset_index(drop=True).iterrows(), desc=\"Tokens\", total=len(df)\n    ):\n        row_fts = fts[row.id]\n\n        inputs = TOKENIZER.encode_plus(\n            row.source,\n            None,\n            add_special_tokens=True,\n            max_length=MD_MAX_LEN,\n            # padding=\"max_length\",\n            return_token_type_ids=True,\n            truncation=True,\n        )\n\n        code_inputs = TOKENIZER.batch_encode_plus(\n            [str(x) for x in row_fts[\"codes\"]] or [\"\"],\n            add_special_tokens=False,\n            max_length=23,\n            # padding=\"max_length\",\n            truncation=True,\n        )\n\n        ids = inputs[\"input_ids\"]\n        for x in code_inputs[\"input_ids\"]:\n            ids.extend(x)\n        ids = ids[:TOTAL_MAX_LEN]\n        ids[-1] = TOKENIZER.sep_token_id  # đủ max len\n        if len(ids) != TOTAL_MAX_LEN:\n            ids = ids + [TOKENIZER.pad_token_id, ] * (TOTAL_MAX_LEN - len(ids))\n\n        mask = inputs[\"attention_mask\"]\n        for x in code_inputs[\"attention_mask\"]:\n            mask.extend(x)\n        mask = mask[:TOTAL_MAX_LEN]\n        mask[-1] = 1\n        if len(mask) != TOTAL_MAX_LEN:\n            mask = mask + [0, ] * (TOTAL_MAX_LEN - len(mask))\n\n        n_md = row_fts[\"total_md\"]\n        n_code = row_fts[\"total_code\"]\n        if n_md + n_code == 0:\n            fts_sum = 0\n        else:\n            fts_sum = n_md / (n_md + n_code)\n\n        assert len(ids) == TOTAL_MAX_LEN\n\n        input_ids[i] = ids\n        attention_mask[i] = mask\n        features[i] = fts_sum\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_mask\": attention_mask,\n        \"features\": features,\n    }\n\ndef get_ranks(base: pd.Series, derived: List[str]) -> List[str]:\n    return [base.index(d) for d in derived]\n\n\ndef get_dataset(\n    input_ids: np.array,\n    attention_mask: np.array,\n    feature: np.array,\n) -> tf.data.Dataset:\n    dataset = tf.data.Dataset.from_tensor_slices(\n        {\"input_ids\": input_ids, \"attention_mask\": attention_mask, \"feature\": feature}\n    )\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset.prefetch(tf.data.AUTOTUNE)\n\n\ndef get_model() -> tf.keras.Model:\n    backbone = transformers.TFAutoModel.from_pretrained(BASE_MODEL)\n    input_ids = tf.keras.layers.Input(\n        shape=(TOTAL_MAX_LEN,),\n        dtype=tf.int32,\n        name=\"input_ids\",\n    )\n    attention_mask = tf.keras.layers.Input(\n        shape=(TOTAL_MAX_LEN,),\n        dtype=tf.int32,\n        name=\"attention_mask\",\n    )\n    feature = tf.keras.layers.Input(\n        shape=(1,),\n        dtype=tf.float32,\n        name=\"feature\",\n    )\n    x = backbone({\"input_ids\": input_ids, \"attention_mask\": attention_mask})[0]\n    x = tf.concat([x[:, 0, :], feature], axis=1)\n    outputs = tf.keras.layers.Dense(1, activation=\"linear\", dtype=\"float32\")(x)\n    return tf.keras.Model(\n        inputs=[input_ids, attention_mask, feature],\n        outputs=outputs,\n    )\n\n######################## process function\nimport markdown\nimport re\n\nTAG_RE = re.compile(r'<[^>]+>')\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*',\n          '+', '\\\\', '•', '~', '@', '£',\n          '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', 'Â', '█',\n          '½', 'à', '…', '\\n', '\\xa0', '\\t',\n          '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥',\n          '▓', '—', '‹', '─', '\\u3000', '\\u202f',\n          '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾',\n          'Ã', '⋅', '‘', '∞', '«',\n          '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹',\n          '≤', '‡', '√', ]\n\n\ndef remove_tags(text):\n    return TAG_RE.sub('', text)\n\n\ndef clean_text(text):\n    html_text = markdown.markdown(text)\n    text_origin = remove_tags(html_text)\n    return text_origin\n\n\ndef clean_character(text):\n    for punct in puncts:\n        text = text.replace(punct, f' {punct} ')\n    return text\n\n\ndef clean_endline(text):\n    return text.replace(\"\\\\n\", \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:25:56.732823Z","iopub.execute_input":"2022-07-17T14:25:56.733792Z","iopub.status.idle":"2022-07-17T14:25:56.769898Z","shell.execute_reply.started":"2022-07-17T14:25:56.733747Z","shell.execute_reply":"2022-07-17T14:25:56.768703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = glob.glob(os.path.join(INPUT_PATH, \"test\", \"*.json\"))\ndf = (\n    pd.concat([read_notebook(x) for x in tqdm(paths, desc=\"Concat\")])\n    .set_index(\"id\", append=True)\n    .swaplevel()\n    .sort_index(level=\"id\", sort_remaining=False)\n).reset_index()\n\ndf[\"source\"] = df[\"source\"].str.slice(0, MD_MAX_LEN)\ndf[\"rank\"] = df.groupby([\"id\", \"cell_type\"]).cumcount()\ndf[\"pct_rank\"] = df.groupby([\"id\", \"cell_type\"])[\"rank\"].rank(pct=True)\n\n# process data in here\nprint(\">> Processing eliminate ...\")\ndf.loc[df[df['cell_type'] == 'markdown'].source.index, 'source'] = df[df['cell_type'] == 'markdown'].source.apply(lambda x: clean_text(x))\ndf.loc[df[df['cell_type'] == 'markdown'].source.index, 'source'] = df[df['cell_type'] == 'markdown'].source.apply(lambda x: clean_character(x))\ndf.loc[df[df['cell_type'] == 'markdown'].source.index, 'source'] = df[df['cell_type'] == 'markdown'].source.apply(lambda x: clean_endline(x))\nprint(\">> Processing eliminate done ...\")\n\n########################\n\n\nfts = get_features(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:37:10.115272Z","iopub.execute_input":"2022-07-17T14:37:10.115646Z","iopub.status.idle":"2022-07-17T14:37:10.258964Z","shell.execute_reply.started":"2022-07-17T14:37:10.115606Z","shell.execute_reply":"2022-07-17T14:37:10.257964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:37:11.292493Z","iopub.execute_input":"2022-07-17T14:37:11.293203Z","iopub.status.idle":"2022-07-17T14:37:11.320436Z","shell.execute_reply.started":"2022-07-17T14:37:11.293165Z","shell.execute_reply":"2022-07-17T14:37:11.319472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with STRATEGY.scope():\n    model = get_model()\n    model.load_weights(\"../input/codebert-processed-v1/model_0_codebert.h5\")\n\npredict = np.array([], dtype=np.float32)\n\nfor chunk in tqdm(\n    np.array_split(df[df[\"cell_type\"] == \"markdown\"], SLICES), total=SLICES\n):\n    if chunk.empty:\n        continue\n\n    data = tokenize(chunk, fts)\n\n    dataset = get_dataset(data[\"input_ids\"], data[\"attention_mask\"], data[\"features\"])\n    predict = np.r_[\n        predict,\n        model.predict(dataset).reshape(\n            -1,\n        ),\n    ]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:37:16.162471Z","iopub.execute_input":"2022-07-17T14:37:16.162865Z","iopub.status.idle":"2022-07-17T14:37:31.681675Z","shell.execute_reply.started":"2022-07-17T14:37:16.162831Z","shell.execute_reply":"2022-07-17T14:37:31.680699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df[\"cell_type\"] == \"markdown\", \"pct_rank\"] = predict\ndf = df.sort_values(\"pct_rank\").groupby(\"id\")[\"cell_id\"].apply(\" \".join)\ndf.name = \"cell_order\"\ndf.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T14:37:31.683754Z","iopub.execute_input":"2022-07-17T14:37:31.684390Z","iopub.status.idle":"2022-07-17T14:37:31.694696Z","shell.execute_reply.started":"2022-07-17T14:37:31.684352Z","shell.execute_reply":"2022-07-17T14:37:31.693788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}