{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Is Fast tokenizer really fast?\n\n### Base check options\n\n```\nclass BaseCFG:\n    epochs = 3\n    num_workers = 2\n    batch_size = 8\n    max_len = 512\n    print_freq = 500\n    tokenizer_name = \"microsoft/deberta-large\"\n```\n\n### How check tokenizer (fast and usual)\n\n1. Load tokenizer\n\n```\nclass CFG(BaseCFG):\n    is_fast = True # OR False\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.tokenizer_name, use_fast=CFG.is_fast)\n```\n\n2. Check status\n\n```\nCFG.tokenizer.is_fast\n```\n\n3. Check runtime\n\n```\n%%time\ntrain_loop(df)\n```\n\n### Additional check (without / with collate_fn)\n\n```\ncollate_fn = Collate(CFG.tokenizer)\n\n%%time\ntrain_loop(df, collate_fn)\n```","metadata":{}},{"cell_type":"markdown","source":"# 1. Import & Def & Set & Load","metadata":{}},{"cell_type":"code","source":"import os\nimport time\nimport math\n\nimport numpy as np\nimport pandas as pd\n\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\n\nfrom transformers import AutoTokenizer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T00:06:43.080185Z","iopub.execute_input":"2022-07-27T00:06:43.081393Z","iopub.status.idle":"2022-07-27T00:06:45.914297Z","shell.execute_reply.started":"2022-07-27T00:06:43.081270Z","shell.execute_reply":"2022-07-27T00:06:45.912978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_essay(essay_id, txt_dir='train'):\n    essay_path = os.path.join(INPUT_DIR + txt_dir, essay_id + '.txt')\n    essay_text = open(essay_path, 'r').read()\n    \n    return essay_text\n\n\ndef asMinutes(s):\n    m = math.floor(s / 60)\n    s -= m * 60\n    \n    return '%dm %ds' % (m, s)\n\n\ndef timeSince(since, percent):\n    now = time.time()\n    s = now - since\n    es = s / (percent)\n    rs = es - s\n    \n    return '%s (remain %s)' % (asMinutes(s), asMinutes(rs))\n    \n    \ndef train_fn(train_loader, device, epoch):\n    start = end = time.time()\n    \n    for step, inputs in enumerate(train_loader):\n        for k, v in inputs.items():\n            inputs[k] = v.to(device)\n            \n        end = time.time()\n\n        if step % CFG.print_freq == 0 or step == (len(train_loader)-1):\n                print('Epoch: [{0}][{1}/{2}] '\n                      'Elapsed {remain:s}'\n                      .format(epoch, step, len(train_loader), \n                              remain=timeSince(start, float(step+1)/len(train_loader)))\n                )\n            \n            \ndef train_loop(df, collate_fn=None):\n    train_dataset = CustomDataset(CFG, df)\n\n    train_loader = DataLoader(\n        train_dataset,\n        batch_size=CFG.batch_size,\n        drop_last=False,\n        shuffle=False,\n        num_workers=CFG.num_workers,\n        pin_memory=True,\n        collate_fn=collate_fn\n    )\n    \n    for epoch in range(1, CFG.epochs + 1):\n        train_fn(train_loader, DEVICE, epoch)\n        \n    print('The End...\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:06:45.916976Z","iopub.execute_input":"2022-07-27T00:06:45.919174Z","iopub.status.idle":"2022-07-27T00:06:45.936333Z","shell.execute_reply.started":"2022-07-27T00:06:45.919124Z","shell.execute_reply":"2022-07-27T00:06:45.934876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BaseCFG:\n    epochs = 3\n    num_workers = 2\n    batch_size = 8\n    max_len = 512\n    print_freq = 500\n    tokenizer_name = \"microsoft/deberta-large\"\n    \nclass CustomDataset(Dataset):\n    def __init__(self, cfg, df):\n        self.cfg = cfg\n        self.texts = df['text'].values\n        \n    def __len__(self):\n        return len(self.texts)\n    \n    def __getitem__(self, item):\n        inputs = prepare_input(self.cfg, self.texts[item])\n        \n        return inputs","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:06:45.937817Z","iopub.execute_input":"2022-07-27T00:06:45.938170Z","iopub.status.idle":"2022-07-27T00:06:45.950892Z","shell.execute_reply.started":"2022-07-27T00:06:45.938139Z","shell.execute_reply":"2022-07-27T00:06:45.949635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', 100)\n\nINPUT_DIR = \"../input/feedback-prize-effectiveness/\"\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:06:45.953652Z","iopub.execute_input":"2022-07-27T00:06:45.954153Z","iopub.status.idle":"2022-07-27T00:06:45.967926Z","shell.execute_reply.started":"2022-07-27T00:06:45.954110Z","shell.execute_reply":"2022-07-27T00:06:45.966883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_origin = pd.read_csv(INPUT_DIR + 'train.csv')\ntrain_origin.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:06:45.968936Z","iopub.execute_input":"2022-07-27T00:06:45.969286Z","iopub.status.idle":"2022-07-27T00:06:46.322517Z","shell.execute_reply.started":"2022-07-27T00:06:45.969257Z","shell.execute_reply":"2022-07-27T00:06:46.320923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_origin.copy()\n\ndf['essay_text'] = df['essay_id'].apply(fetch_essay)\ndf['text'] = df['discourse_type'] + ' ' + df['discourse_text'] + '[SEP]' + df['essay_text']\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:06:46.324994Z","iopub.execute_input":"2022-07-27T00:06:46.325327Z","iopub.status.idle":"2022-07-27T00:07:16.379672Z","shell.execute_reply.started":"2022-07-27T00:06:46.325289Z","shell.execute_reply":"2022-07-27T00:07:16.378267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Without collate_fn","metadata":{}},{"cell_type":"code","source":"def prepare_input(cfg, text):\n    inputs = cfg.tokenizer(text,\n                           add_special_tokens=True,\n                           max_length=cfg.max_len,\n                           padding=\"max_length\",\n                           truncation=True)\n\n    for k, v in inputs.items():\n        inputs[k] = torch.tensor(v, dtype=torch.long)\n        \n    return inputs","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:07:16.381646Z","iopub.execute_input":"2022-07-27T00:07:16.381984Z","iopub.status.idle":"2022-07-27T00:07:16.388196Z","shell.execute_reply.started":"2022-07-27T00:07:16.381953Z","shell.execute_reply":"2022-07-27T00:07:16.387123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1 Usual tokenizer","metadata":{}},{"cell_type":"code","source":"class CFG(BaseCFG):\n    is_fast = False\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.tokenizer_name, use_fast=CFG.is_fast)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:07:16.389608Z","iopub.execute_input":"2022-07-27T00:07:16.389920Z","iopub.status.idle":"2022-07-27T00:07:23.262996Z","shell.execute_reply.started":"2022-07-27T00:07:16.389890Z","shell.execute_reply":"2022-07-27T00:07:23.261946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.tokenizer.is_fast","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:07:23.265788Z","iopub.execute_input":"2022-07-27T00:07:23.266922Z","iopub.status.idle":"2022-07-27T00:07:23.275129Z","shell.execute_reply.started":"2022-07-27T00:07:23.266873Z","shell.execute_reply":"2022-07-27T00:07:23.273843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_loop(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:07:23.276669Z","iopub.execute_input":"2022-07-27T00:07:23.277090Z","iopub.status.idle":"2022-07-27T00:14:10.056258Z","shell.execute_reply.started":"2022-07-27T00:07:23.277057Z","shell.execute_reply":"2022-07-27T00:14:10.055259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Fast tokenizer","metadata":{}},{"cell_type":"code","source":"class CFG(BaseCFG):\n    is_fast = True\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.tokenizer_name, use_fast=CFG.is_fast)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:14:10.057922Z","iopub.execute_input":"2022-07-27T00:14:10.058264Z","iopub.status.idle":"2022-07-27T00:14:15.458781Z","shell.execute_reply.started":"2022-07-27T00:14:10.058232Z","shell.execute_reply":"2022-07-27T00:14:15.457742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.tokenizer.is_fast","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:14:15.460290Z","iopub.execute_input":"2022-07-27T00:14:15.461191Z","iopub.status.idle":"2022-07-27T00:14:15.468605Z","shell.execute_reply.started":"2022-07-27T00:14:15.461145Z","shell.execute_reply":"2022-07-27T00:14:15.467642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_loop(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:14:15.469855Z","iopub.execute_input":"2022-07-27T00:14:15.470562Z","iopub.status.idle":"2022-07-27T00:17:06.234041Z","shell.execute_reply.started":"2022-07-27T00:14:15.470529Z","shell.execute_reply":"2022-07-27T00:17:06.232625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. With collate_fn","metadata":{}},{"cell_type":"code","source":"def prepare_input(cfg, text):\n    inputs = cfg.tokenizer(text,\n                           add_special_tokens=True,\n                           padding=True,\n                           truncation=True)\n        \n    return inputs\n\n\nclass Collate:\n    def __init__(self, tokenizer):\n        self.tokenizer = tokenizer\n\n    def __call__(self, batch):\n        output = dict()\n        output[\"input_ids\"] = [sample[\"input_ids\"] for sample in batch]\n        output[\"attention_mask\"] = [sample[\"attention_mask\"] for sample in batch]\n        \n        # calculate max token length of this batch\n        batch_max = max([len(ids) for ids in output[\"input_ids\"]])\n\n        # add padding\n        if self.tokenizer.padding_side == \"right\":\n            output[\"input_ids\"] = [s + (batch_max - len(s)) * [self.tokenizer.pad_token_id] for s in output[\"input_ids\"]]\n            output[\"attention_mask\"] = [s + (batch_max - len(s)) * [0] for s in output[\"attention_mask\"]]\n        else:\n            output[\"input_ids\"] = [(batch_max - len(s)) * [self.tokenizer.pad_token_id] + s for s in output[\"input_ids\"]]\n            output[\"attention_mask\"] = [(batch_max - len(s)) * [0] + s for s in output[\"attention_mask\"]]\n\n        # convert to tensors\n        output[\"input_ids\"] = torch.tensor(output[\"input_ids\"], dtype=torch.long)\n        output[\"attention_mask\"] = torch.tensor(output[\"attention_mask\"], dtype=torch.long)\n\n        return output","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:17:06.235917Z","iopub.execute_input":"2022-07-27T00:17:06.236315Z","iopub.status.idle":"2022-07-27T00:17:06.252557Z","shell.execute_reply.started":"2022-07-27T00:17:06.236281Z","shell.execute_reply":"2022-07-27T00:17:06.251631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1 Usual tokenizer","metadata":{}},{"cell_type":"code","source":"class CFG(BaseCFG):\n    is_fast = False\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.tokenizer_name, use_fast=CFG.is_fast)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:17:06.256649Z","iopub.execute_input":"2022-07-27T00:17:06.257685Z","iopub.status.idle":"2022-07-27T00:17:10.132508Z","shell.execute_reply.started":"2022-07-27T00:17:06.257639Z","shell.execute_reply":"2022-07-27T00:17:10.131383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.tokenizer.is_fast","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:17:10.134282Z","iopub.execute_input":"2022-07-27T00:17:10.134604Z","iopub.status.idle":"2022-07-27T00:17:10.141894Z","shell.execute_reply.started":"2022-07-27T00:17:10.134573Z","shell.execute_reply":"2022-07-27T00:17:10.140399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_fn = Collate(CFG.tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:17:10.143226Z","iopub.execute_input":"2022-07-27T00:17:10.143551Z","iopub.status.idle":"2022-07-27T00:17:10.151332Z","shell.execute_reply.started":"2022-07-27T00:17:10.143510Z","shell.execute_reply":"2022-07-27T00:17:10.150136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_loop(df, collate_fn)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:17:10.153282Z","iopub.execute_input":"2022-07-27T00:17:10.153825Z","iopub.status.idle":"2022-07-27T00:23:36.122249Z","shell.execute_reply.started":"2022-07-27T00:17:10.153790Z","shell.execute_reply":"2022-07-27T00:23:36.121137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2 Fast tokenizer","metadata":{}},{"cell_type":"code","source":"class CFG(BaseCFG):\n    is_fast = True\n\nCFG.tokenizer = AutoTokenizer.from_pretrained(CFG.tokenizer_name, use_fast=CFG.is_fast)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:23:36.124147Z","iopub.execute_input":"2022-07-27T00:23:36.124532Z","iopub.status.idle":"2022-07-27T00:23:41.523965Z","shell.execute_reply.started":"2022-07-27T00:23:36.124495Z","shell.execute_reply":"2022-07-27T00:23:41.522775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.tokenizer.is_fast","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:23:41.525428Z","iopub.execute_input":"2022-07-27T00:23:41.525809Z","iopub.status.idle":"2022-07-27T00:23:41.533482Z","shell.execute_reply.started":"2022-07-27T00:23:41.525777Z","shell.execute_reply":"2022-07-27T00:23:41.532212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_fn = Collate(CFG.tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:23:41.534982Z","iopub.execute_input":"2022-07-27T00:23:41.535299Z","iopub.status.idle":"2022-07-27T00:23:41.548712Z","shell.execute_reply.started":"2022-07-27T00:23:41.535271Z","shell.execute_reply":"2022-07-27T00:23:41.547539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_loop(df, collate_fn)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:23:41.550523Z","iopub.execute_input":"2022-07-27T00:23:41.551646Z","iopub.status.idle":"2022-07-27T00:26:18.918301Z","shell.execute_reply.started":"2022-07-27T00:23:41.551601Z","shell.execute_reply":"2022-07-27T00:26:18.916839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}