{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!pip install bnunicodenormalizer --no-index --find-links=file:///kaggle/input/environment/\n\n#!pip install pyctcdecode --no-index --find-links=file:///kaggle/input/dependencyies/\n\n!pip install wordfreq --no-index --find-links=file:///kaggle/input/environment/\n\n!pip install symspellpy --no-index --find-links=file:///kaggle/input/environment/\n#!pip install /kaggle/input/dependencyies/master.zip\n\n!pip install pyctcdecode --no-index --find-links=file:///kaggle/input/environment/\n\n!pip install conda-pack --no-index --find-links=file:///kaggle/input/environment/\n\n!pip install /kaggle/input/environment/master.zip\n\n!sudo mkdir -p /root/.cache/torch/transformers/\n\n!sudo cp -r /kaggle/input/punctuation-environment/transformers/* /root/.cache/torch/transformers/\n\n#!conda run pip install conda-pack --no-index --find-links=file:///kaggle/input/environment/\n!conda init bash\n\n!mkdir -p /kaggle/working/punctuation\n!tar -xzf /kaggle/input/punctuation-environment/punctuation.tar.gz -C /kaggle/working/punctuation\n!source /kaggle/working/punctuation/bin/activate\n\n!conda run --prefix=/kaggle/working/punctuation python  -c \"import transformers;print(transformers.__version__)\"\n\n%cd /kaggle/working/punctuation/punctuation-restoration/","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:23:27.102937Z","iopub.execute_input":"2022-08-30T02:23:27.103265Z","iopub.status.idle":"2022-08-30T02:29:06.045370Z","shell.execute_reply.started":"2022-08-30T02:23:27.103191Z","shell.execute_reply":"2022-08-30T02:29:06.044211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sed -i 's/print(result)//g' src/inference.py","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:31:56.062143Z","iopub.execute_input":"2022-08-30T02:31:56.063112Z","iopub.status.idle":"2022-08-30T02:31:57.097972Z","shell.execute_reply.started":"2022-08-30T02:31:56.063074Z","shell.execute_reply":"2022-08-30T02:31:57.096622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 16\nTEST_DIRECTORY = '/kaggle/input/dlsprint/test_files'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-30T02:29:06.048834Z","iopub.execute_input":"2022-08-30T02:29:06.049181Z","iopub.status.idle":"2022-08-30T02:29:06.055008Z","shell.execute_reply.started":"2022-08-30T02:29:06.049147Z","shell.execute_reply":"2022-08-30T02:29:06.053777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Please refactor your inference code into the below functions. It doesn't matter which function you use to ultimately infer on the test set, you can use any one. But be sure to implement working versions of these function formats.\n\nimport numpy as np\nimport pandas as pd\nimport random\nimport ast\nfrom tqdm import tqdm\nfrom IPython import display as ipd\n\n# visualization\nimport matplotlib.pyplot as plt\nfrom tabulate import tabulate\nimport seaborn as sns\n#system files\nimport os\nimport json\nimport re\nimport glob\n\nimport zipfile\nimport shutil\nimport gc\nfrom pydub import AudioSegment\nfrom joblib import Parallel, delayed\n\nfrom transformers import (AutoTokenizer,\n                          AutoFeatureExtractor,\n                          AutoConfig,\n                          AutoModel,\n                          Wav2Vec2CTCTokenizer,\n                          Wav2Vec2ForCTC,\n                          Wav2Vec2Processor,\n                          Trainer,\n                          TrainingArguments,\n                          Wav2Vec2FeatureExtractor,\n                          get_linear_schedule_with_warmup,\n                          set_seed,\n                          Wav2Vec2ProcessorWithLM)\n\n\n\n\n# PyTorch \nimport torch\nimport torchaudio\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torch.nn.functional as F\nimport torchaudio.functional as FT\nimport torchaudio.transforms as TT\n\n\n#sklearn\nfrom sklearn.model_selection import train_test_split\nfrom datasets import (load_dataset,\n                      load_metric,\n                      Dataset,\n                      concatenate_datasets,\n                      set_caching_enabled,\n                      ClassLabel,\n                      Audio)\n\nimport librosa\n\n\nfrom wordfreq import (word_frequency,\n                      top_n_list,\n                      get_frequency_dict,\n                      zipf_frequency)\n\nfrom symspellpy import SymSpell, Verbosity\nfrom itertools import islice\nfrom collections import Counter\n\nfrom pandarallel import pandarallel\nfrom bnunicodenormalizer import Normalizer \npandarallel.initialize(progress_bar=True,nb_workers=8)\ntqdm.pandas()\nbnorm=Normalizer()\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:06.056715Z","iopub.execute_input":"2022-08-30T02:29:06.057528Z","iopub.status.idle":"2022-08-30T02:29:16.412723Z","shell.execute_reply.started":"2022-08-30T02:29:06.057482Z","shell.execute_reply":"2022-08-30T02:29:16.411577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"WANDB_DISABLED\"] = \"true\"","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:16.415656Z","iopub.execute_input":"2022-08-30T02:29:16.416998Z","iopub.status.idle":"2022-08-30T02:29:16.422883Z","shell.execute_reply.started":"2022-08-30T02:29:16.416965Z","shell.execute_reply":"2022-08-30T02:29:16.421831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# link to this dataset and how it was created is in part_3_bangla_dictionary notebook","metadata":{}},{"cell_type":"code","source":"sym_spell_word_segmentation = SymSpell(max_dictionary_edit_distance=0, prefix_length=7)\ndictionary_path = '/kaggle/input/bangla-frequency-dictionary/symspell.txt'\nsym_spell_word_segmentation.load_dictionary(dictionary_path, 0, 1,separator=\",\")","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:16.424685Z","iopub.execute_input":"2022-08-30T02:29:16.425047Z","iopub.status.idle":"2022-08-30T02:29:19.689117Z","shell.execute_reply.started":"2022-08-30T02:29:16.425011Z","shell.execute_reply":"2022-08-30T02:29:19.688047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sym_spell = SymSpell(max_dictionary_edit_distance=5, prefix_length=7)\ndictionary_path = '/kaggle/input/bangla-frequency-dictionary/symspell.txt'\nsym_spell.load_dictionary(dictionary_path, 0, 1,separator=\",\")","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:19.691687Z","iopub.execute_input":"2022-08-30T02:29:19.692762Z","iopub.status.idle":"2022-08-30T02:29:49.129681Z","shell.execute_reply.started":"2022-08-30T02:29:19.692725Z","shell.execute_reply":"2022-08-30T02:29:49.128767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():  \n    device = \"cuda:0\" \nelse:  \n    device = \"cpu\"  ","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.130958Z","iopub.execute_input":"2022-08-30T02:29:49.131602Z","iopub.status.idle":"2022-08-30T02:29:49.166552Z","shell.execute_reply.started":"2022-08-30T02:29:49.131574Z","shell.execute_reply":"2022-08-30T02:29:49.165484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.168229Z","iopub.execute_input":"2022-08-30T02:29:49.168995Z","iopub.status.idle":"2022-08-30T02:29:49.178431Z","shell.execute_reply.started":"2022-08-30T02:29:49.168957Z","shell.execute_reply":"2022-08-30T02:29:49.176975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This model is hosted on huggingface detail is given in part_2_wav2vec2_language_model and part_1_wav2vec2_model_training","metadata":{}},{"cell_type":"code","source":"model_path='/kaggle/input/environment-and-variables-setup/wav2vec2/'","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.179953Z","iopub.execute_input":"2022-08-30T02:29:49.180353Z","iopub.status.idle":"2022-08-30T02:29:49.186686Z","shell.execute_reply.started":"2022-08-30T02:29:49.180318Z","shell.execute_reply":"2022-08-30T02:29:49.185622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def word_segmentation(input_term):\n    result = sym_spell_word_segmentation.word_segmentation(input_term)\n    return result.corrected_string","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.190351Z","iopub.execute_input":"2022-08-30T02:29:49.190840Z","iopub.status.idle":"2022-08-30T02:29:49.197581Z","shell.execute_reply.started":"2022-08-30T02:29:49.190806Z","shell.execute_reply":"2022-08-30T02:29:49.196456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dictionary_(word):\n    suggestions = sym_spell.lookup(\n    word, Verbosity.CLOSEST,max_edit_distance=2, include_unknown=True)\n    for suggestion in suggestions:\n        return str(suggestion).split(',')[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.199078Z","iopub.execute_input":"2022-08-30T02:29:49.199676Z","iopub.status.idle":"2022-08-30T02:29:49.207330Z","shell.execute_reply.started":"2022-08-30T02:29:49.199641Z","shell.execute_reply":"2022-08-30T02:29:49.206313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lookup(sen):\n    words = sen.split()\n    m=[]\n    for wow in words:\n        if len(wow)>16:\n            s=word_segmentation(wow)\n            j = s.split()\n            for n in j:\n                m.append(n)\n        else:\n            m.append(wow)\n    l=[]\n    for wow in m:\n        if (word_frequency(wow,'bn',wordlist='large',minimum=0.0) == 0.0):\n            s=dictionary_(wow)\n            l.append(s)\n        else:\n            l.append(wow)\n    return ' '.join(l)","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.210339Z","iopub.execute_input":"2022-08-30T02:29:49.211037Z","iopub.status.idle":"2022-08-30T02:29:49.219153Z","shell.execute_reply.started":"2022-08-30T02:29:49.211002Z","shell.execute_reply":"2022-08-30T02:29:49.217918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cleaning_punctuation(sen):\n    if len(sen) != 0:\n        sen=sen.strip()\n        all_words=[]\n        for word in sen.split()[:-1]:\n            if '।' in word:\n                all_words.append(word[:-1]+',')\n            else:\n                all_words.append(word)\n        final_word = sen.split()[-1]\n        if final_word[-1] != '।':\n            if final_word[-1] == '?':\n                all_words.append(final_word)\n            if final_word[-1] == ',':\n                all_words.append(final_word[:-1]+'।')\n            if final_word[-1] not in [',','?']:\n                all_words.append(final_word+'।')\n        else:\n            all_words.append(final_word)\n    else:\n        return sen\n    return \" \".join(all_words)","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:29:49.220617Z","iopub.execute_input":"2022-08-30T02:29:49.220972Z","iopub.status.idle":"2022-08-30T02:29:49.230381Z","shell.execute_reply.started":"2022-08-30T02:29:49.220940Z","shell.execute_reply":"2022-08-30T02:29:49.229112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def punctuation_alignment(sen):\n    with open('data/output_bn.txt') as f:\n        lines = f.readlines()\n    lines = lines[0].split()\n    if isinstance(sen, str):\n        sen = sen.strip()\n        l = len(sen)\n        if l != 0:\n            sen = lines[:l]\n            return \" \".join(sen)\n        else:\n            return sen\n    if isinstance(sen, pd.core.frame.DataFrame):\n        df=sen\n        nan_id=df.loc[pd.isna(df[\"sentence\"]), :].index\n        for i in range(len(df)-len(nan_id)):\n            line_p = df.path[i]\n            line_g = df.sentence[i].split()\n            l = len(line_g)\n            sentence=lines[:l]\n            lines=lines[l:]\n            sentence = \" \".join(sentence)\n            df[\"sentence\"][i] = sentence\n        return df","metadata":{"execution":{"iopub.status.busy":"2022-08-30T03:24:03.989008Z","iopub.execute_input":"2022-08-30T03:24:03.989469Z","iopub.status.idle":"2022-08-30T03:24:03.998904Z","shell.execute_reply.started":"2022-08-30T03:24:03.989422Z","shell.execute_reply":"2022-08-30T03:24:03.997874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_processing(sen):\n    if isinstance(sen, str):\n        spelling = lookup(sen)\n        final = punctuation(spelling)\n        punctuation = punctuation_cleaner()\n        return final\n    if isinstance(sen, list):\n        spelling=[lookup(item) for item in sen]\n        final=[punctuation(item) for item in spelling]\n        return final\n    else:\n        return sen","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:31:42.618291Z","iopub.execute_input":"2022-08-30T02:31:42.618657Z","iopub.status.idle":"2022-08-30T02:31:42.626845Z","shell.execute_reply.started":"2022-08-30T02:31:42.618623Z","shell.execute_reply":"2022-08-30T02:31:42.623912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer on a single data","metadata":{}},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(model_path).to(device)\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(model_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:32:07.911104Z","iopub.execute_input":"2022-08-30T02:32:07.911576Z","iopub.status.idle":"2022-08-30T02:32:31.248799Z","shell.execute_reply.started":"2022-08-30T02:32:07.911531Z","shell.execute_reply":"2022-08-30T02:32:31.247796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(audio_path):\n    df_submission = pd.DataFrame()\n    dictionary = {'path':audio_path}\n    df_submission = df_submission.append(dictionary, ignore_index = True)\n    submission = Dataset.from_pandas(df_submission)\n    submission = submission.cast_column(\"path\", Audio(sampling_rate=16_000))\n    #model = Wav2Vec2ForCTC.from_pretrained(model_path).to(device)\n    #processor = Wav2Vec2ProcessorWithLM.from_pretrained(model_path)\n    inputs = processor(submission[0][\"path\"][\"array\"], sampling_rate=16_000, return_tensors=\"pt\").to(device)\n    with torch.no_grad():\n        logits = model(**inputs).logits\n    transcription = processor.batch_decode(logits.cpu().numpy()).text\n    #del model\n    sen=transcription[0]\n    sen = lookup(sen)\n    torch.cuda.empty_cache()\n    gc.collect()\n    #sen = post_processing(transcription[0])\n    if os.path.isfile('data/input_bn.txt'):\n        os.remove('data/input_bn.txt')\n    with open('data/input_bn.txt', 'w') as f:\n        f.write(sen)\n    return sen","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:32:42.823462Z","iopub.execute_input":"2022-08-30T02:32:42.823917Z","iopub.status.idle":"2022-08-30T02:32:42.835554Z","shell.execute_reply.started":"2022-08-30T02:32:42.823877Z","shell.execute_reply":"2022-08-30T02:32:42.834526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#path='/kaggle/input/dlsprint/train_files/common_voice_bn_30648521.mp3'","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:32:44.488433Z","iopub.execute_input":"2022-08-30T02:32:44.488797Z","iopub.status.idle":"2022-08-30T02:32:44.494075Z","shell.execute_reply.started":"2022-08-30T02:32:44.488763Z","shell.execute_reply":"2022-08-30T02:32:44.493083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sen=infer(path)\n#!conda run -p \"/kaggle/working/punctuation\" python src/inference.py --pretrained-model=xlm-roberta-large --weight-path=/kaggle/input/environment-and-variables-setup/punctuation_adder/xlm-roberta-large-bn.pt --language=bn --in-file=data/input_bn.txt --out-file=data/output_bn.txt\n#sen=punctuation_alignment(sen)\n#sen=cleaning_punctuation(sen)\n#sen","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:32:48.347432Z","iopub.execute_input":"2022-08-30T02:32:48.347918Z","iopub.status.idle":"2022-08-30T02:35:00.143110Z","shell.execute_reply.started":"2022-08-30T02:32:48.347870Z","shell.execute_reply":"2022-08-30T02:35:00.141901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer on a batch of data - MOST IMPORTANT","metadata":{}},{"cell_type":"code","source":"def batch_infer(audio_paths, batch_size=BATCH_SIZE):\n    df_submission = pd.DataFrame()\n    df_submission['path'] = audio_paths\n    submission = Dataset.from_pandas(df_submission)\n    path = \"/kaggle/tmp/submission\"\n    isExist = os.path.exists(path)\n    if not isExist:\n        os.makedirs(path)\n    submission.save_to_disk(path)\n    submission = submission.cast_column(\"path\", Audio(sampling_rate=16_000))\n    submission.cleanup_cache_files()\n    gc.collect()\n    result=[]\n    for i in tqdm(range(len(submission))):\n        inputs = processor(submission[i][\"path\"][\"array\"], sampling_rate=16_000, return_tensors=\"pt\").to(device)\n        with torch.no_grad():\n            logits = model(**inputs).logits\n            transcription = processor.batch_decode(logits.cpu().numpy()).text\n            result.append(transcription[0])\n    torch.cuda.empty_cache()\n    sen=[lookup(item) for item in result]\n    if os.path.isfile('data/input_bn.txt'):\n        os.remove('data/input_bn.txt')\n    textfile = open(\"data/input_bn.txt\", \"w\")\n    for element in sen:\n        textfile.write(element + \"\\n\")\n    textfile.close()\n    return result \n    \n    #infers on a batch of audio\n    #args:\n    #  audio_paths  : list of path to audio files <list of string>\n    #returns:\n    #  bangla predicted texts <list of string>\n","metadata":{"execution":{"iopub.status.busy":"2022-08-30T02:49:20.278149Z","iopub.execute_input":"2022-08-30T02:49:20.278578Z","iopub.status.idle":"2022-08-30T02:49:20.290255Z","shell.execute_reply.started":"2022-08-30T02:49:20.278536Z","shell.execute_reply":"2022-08-30T02:49:20.289137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer on a directory","metadata":{}},{"cell_type":"code","source":"def directory_infer(audio_dir):\n    '''\n    infers on a directory that contains audio files\n    args:\n      audio_dir  : directory that contains some audio files <string>\n    returns:\n      a dataframe that contains 2 columns:\n        * path <string>\n        * sentence <string>\n    '''\n    # list all audio files\n    audio_paths=glob.glob(f\"{audio_dir}/*\")\n    audio_ext = [os.path.splitext(x) for x in audio_paths]\n    audio_paths_actual=[x+y for (x,y) in audio_ext if y in ['.mp3','.wav']]\n    audio_paths_actual=audio_paths_actual[:1000]\n    sentences=batch_infer(audio_paths_actual)\n    #for idx in tqdm(range(0,len(audio_paths),BATCH_SIZE)):\n    #    batch_paths=audio_paths[idx:idx+BATCH_SIZE]\n     #   sentences+=batch_infer(batch_paths)\n        \n    df=pd.DataFrame({\"path\":audio_paths_actual,\"sentence\":sentences})\n    nan_id=df.loc[pd.isna(df[\"sentence\"]), :].index\n    deleted = df[df.index.isin(nan_id)]['sentence'].tolist()\n    df=df.drop(nan_id)\n    df=df.reset_index(drop=True)\n    for i in range(len(deleted)):\n        df = df.append({'path': deleted[i],'sentence':'',},ignore_index=True)\n    return df ","metadata":{"execution":{"iopub.status.busy":"2022-08-30T03:17:08.013333Z","iopub.execute_input":"2022-08-30T03:17:08.013749Z","iopub.status.idle":"2022-08-30T03:17:08.024026Z","shell.execute_reply.started":"2022-08-30T03:17:08.013714Z","shell.execute_reply":"2022-08-30T03:17:08.022925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=directory_infer('/kaggle/input/dlsprint/test_files')\n\n!conda run -p \"/kaggle/working/punctuation\" python src/inference.py --pretrained-model=xlm-roberta-large --weight-path=/kaggle/input/environment-and-variables-setup/punctuation_adder/xlm-roberta-large-bn.pt --language=bn --in-file=data/input_bn.txt --out-file=data/output_bn.txt\n\ndf=punctuation_alignment(submission)\ndf.sentence = df.sentence.apply(lambda x:cleaning_punctuation(str(x)))","metadata":{"execution":{"iopub.status.busy":"2022-08-30T03:26:26.999596Z","iopub.execute_input":"2022-08-30T03:26:27.000021Z","iopub.status.idle":"2022-08-30T03:26:27.132456Z","shell.execute_reply.started":"2022-08-30T03:26:26.999979Z","shell.execute_reply":"2022-08-30T03:26:27.131555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/kaggle/working/inference.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-30T03:26:41.753964Z","iopub.execute_input":"2022-08-30T03:26:41.754728Z","iopub.status.idle":"2022-08-30T03:26:41.768329Z","shell.execute_reply.started":"2022-08-30T03:26:41.754688Z","shell.execute_reply":"2022-08-30T03:26:41.766909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/punctuation","metadata":{"execution":{"iopub.status.busy":"2022-08-30T04:29:04.005451Z","iopub.execute_input":"2022-08-30T04:29:04.005873Z","iopub.status.idle":"2022-08-30T04:29:06.487812Z","shell.execute_reply.started":"2022-08-30T04:29:04.005823Z","shell.execute_reply":"2022-08-30T04:29:06.486413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}