{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install keras-nlp --quiet\n\nimport pandas as pd\nimport keras_nlp as nlp\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-20T17:48:38.741928Z","iopub.execute_input":"2023-08-20T17:48:38.743086Z","iopub.status.idle":"2023-08-20T17:49:08.970913Z","shell.execute_reply.started":"2023-08-20T17:48:38.743041Z","shell.execute_reply":"2023-08-20T17:49:08.969625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_text as text","metadata":{"execution":{"iopub.status.busy":"2023-08-20T18:27:13.968734Z","iopub.execute_input":"2023-08-20T18:27:13.969185Z","iopub.status.idle":"2023-08-20T18:27:13.975736Z","shell.execute_reply.started":"2023-08-20T18:27:13.969153Z","shell.execute_reply":"2023-08-20T18:27:13.973941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = \"/kaggle/input/bengaliai-speech/train.csv\"\ndf = pd.read_csv(file)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T17:49:39.368945Z","iopub.execute_input":"2023-08-20T17:49:39.369846Z","iopub.status.idle":"2023-08-20T17:49:45.385497Z","shell.execute_reply.started":"2023-08-20T17:49:39.369798Z","shell.execute_reply":"2023-08-20T17:49:45.383745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VOCAB_SIZE = 1_000_000\n\nreserved_token = [\"<start>\", \"<end>\", \"<unk>\", \"<pad>\", \"<mask>\"]\n\ndef train_word_piece(text_samples, vocab_size, reserved_tokens):\n    word_piece_ds = tf.data.Dataset.from_tensor_slices(text_samples)\n    vocab = nlp.tokenizers.compute_word_piece_vocabulary(\n        word_piece_ds.batch(1000).prefetch(100),\n        vocabulary_size=vocab_size,\n        reserved_tokens=reserved_tokens,\n    )\n    return vocab\n\nvocab = train_word_piece(\n    text_samples=df[\"sentence\"].to_list(),\n    vocab_size=VOCAB_SIZE, \n    reserved_tokens=reserved_token\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T17:53:52.325605Z","iopub.execute_input":"2023-08-20T17:53:52.326000Z","iopub.status.idle":"2023-08-20T18:13:24.701036Z","shell.execute_reply.started":"2023-08-20T17:53:52.325970Z","shell.execute_reply":"2023-08-20T18:13:24.699567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"vocab size :\", len(vocab))","metadata":{"execution":{"iopub.status.busy":"2023-08-20T18:13:29.282940Z","iopub.execute_input":"2023-08-20T18:13:29.283362Z","iopub.status.idle":"2023-08-20T18:13:29.289575Z","shell.execute_reply.started":"2023-08-20T18:13:29.283330Z","shell.execute_reply":"2023-08-20T18:13:29.288094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(vocab[100:150])","metadata":{"execution":{"iopub.status.busy":"2023-08-20T18:13:32.074905Z","iopub.execute_input":"2023-08-20T18:13:32.075325Z","iopub.status.idle":"2023-08-20T18:13:32.083536Z","shell.execute_reply.started":"2023-08-20T18:13:32.075294Z","shell.execute_reply":"2023-08-20T18:13:32.081868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib","metadata":{"execution":{"iopub.status.busy":"2023-08-20T18:52:10.992741Z","iopub.execute_input":"2023-08-20T18:52:10.993164Z","iopub.status.idle":"2023-08-20T18:52:10.998704Z","shell.execute_reply.started":"2023-08-20T18:52:10.993132Z","shell.execute_reply":"2023-08-20T18:52:10.997418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_vocab(vocab):\n    \"\"\"\n    Saves the processed vocab file as 'vocab.json', to be ingested by tokenizer\n    \"\"\"\n\n    pathlib.Path(\"Vocab.txt\").write_text(\"\\n\".join(vocab))\n    print(\"Created Vocab file!\")\n    \nsave_vocab(vocab)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T18:21:33.834018Z","iopub.execute_input":"2023-08-20T18:21:33.834496Z","iopub.status.idle":"2023-08-20T18:21:33.940888Z","shell.execute_reply.started":"2023-08-20T18:21:33.834446Z","shell.execute_reply":"2023-08-20T18:21:33.939211Z"},"trusted":true},"execution_count":null,"outputs":[]}]}