{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":4143520,"sourceType":"datasetVersion","datasetId":2447262},{"sourceId":6346454,"sourceType":"datasetVersion","datasetId":3654479},{"sourceId":6401888,"sourceType":"datasetVersion","datasetId":3684230},{"sourceId":6553828,"sourceType":"datasetVersion","datasetId":3787222},{"sourceId":6626284,"sourceType":"datasetVersion","datasetId":3825254},{"sourceId":6693882,"sourceType":"datasetVersion","datasetId":3859399},{"sourceId":6713439,"sourceType":"datasetVersion","datasetId":3825965},{"sourceId":6736495,"sourceType":"datasetVersion","datasetId":3812841},{"sourceId":6756124,"sourceType":"datasetVersion","datasetId":3720828},{"sourceId":8164936,"sourceType":"datasetVersion","datasetId":4831307},{"sourceId":8172107,"sourceType":"datasetVersion","datasetId":4836664},{"sourceId":8180720,"sourceType":"datasetVersion","datasetId":4843288},{"sourceId":8195026,"sourceType":"datasetVersion","datasetId":4853938},{"sourceId":8200802,"sourceType":"datasetVersion","datasetId":4858169},{"sourceId":8216339,"sourceType":"datasetVersion","datasetId":4870029},{"sourceId":140325995,"sourceType":"kernelVersion"},{"sourceId":144415834,"sourceType":"kernelVersion"},{"sourceId":144542917,"sourceType":"kernelVersion"},{"sourceId":172803026,"sourceType":"kernelVersion"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"markdown","source":"# Youtube link of methodology\nhttps://youtu.be/Q9VLwuhDGTE","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\nimport os\n!python -m pip install --no-index --find-links=../input/bengaliai-pip-wheels-demucs demucs","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-24T13:44:57.286722Z","iopub.execute_input":"2024-04-24T13:44:57.287317Z","iopub.status.idle":"2024-04-24T13:46:45.154211Z","shell.execute_reply.started":"2024-04-24T13:44:57.287286Z","shell.execute_reply":"2024-04-24T13:46:45.153195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:46:45.156146Z","iopub.execute_input":"2024-04-24T13:46:45.156444Z","iopub.status.idle":"2024-04-24T13:46:46.111412Z","shell.execute_reply.started":"2024-04-24T13:46:45.156415Z","shell.execute_reply":"2024-04-24T13:46:46.110235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir ./generated_audio","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:46:46.112780Z","iopub.execute_input":"2024-04-24T13:46:46.113078Z","iopub.status.idle":"2024-04-24T13:46:47.066286Z","shell.execute_reply.started":"2024-04-24T13:46:46.113051Z","shell.execute_reply":"2024-04-24T13:46:47.065217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference2","metadata":{}},{"cell_type":"code","source":"%%python\nimport typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport re\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:46:47.067883Z","iopub.execute_input":"2024-04-24T13:46:47.068259Z","iopub.status.idle":"2024-04-24T13:46:56.294602Z","shell.execute_reply.started":"2024-04-24T13:46:47.068222Z","shell.execute_reply":"2024-04-24T13:46:56.293787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\nROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = Path('./generated_audio')\n\nSAMPLING_RATE = 16_000\nMODEL_PATH = \"/kaggle/input/ai4bharat-indicwav2vec-v1-bengali/indicwav2vec_v1_bengali/\"   #INPUT / \"bengali-sr-download-public-trained-models/indicwav2vec_v1_bengali/\"\nLM_PATH = \"/kaggle/input/language-model-cd-v1/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:46:56.297102Z","iopub.execute_input":"2024-04-24T13:46:56.297398Z","iopub.status.idle":"2024-04-24T13:46:56.305246Z","shell.execute_reply.started":"2024-04-24T13:46:56.297373Z","shell.execute_reply":"2024-04-24T13:46:56.304147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:46:56.306289Z","iopub.execute_input":"2024-04-24T13:46:56.306587Z","iopub.status.idle":"2024-04-24T13:47:00.119938Z","shell.execute_reply.started":"2024-04-24T13:46:56.306562Z","shell.execute_reply":"2024-04-24T13:47:00.119043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:00.121111Z","iopub.execute_input":"2024-04-24T13:47:00.121526Z","iopub.status.idle":"2024-04-24T13:47:00.145773Z","shell.execute_reply.started":"2024-04-24T13:47:00.121501Z","shell.execute_reply":"2024-04-24T13:47:00.144889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor = Wav2Vec2Processor.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:00.146943Z","iopub.execute_input":"2024-04-24T13:47:00.147225Z","iopub.status.idle":"2024-04-24T13:47:00.221318Z","shell.execute_reply.started":"2024-04-24T13:47:00.147201Z","shell.execute_reply":"2024-04-24T13:47:00.220463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:00.222352Z","iopub.execute_input":"2024-04-24T13:47:00.222611Z","iopub.status.idle":"2024-04-24T13:47:17.698690Z","shell.execute_reply.started":"2024-04-24T13:47:00.222588Z","shell.execute_reply":"2024-04-24T13:47:17.697763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_state_dict(torch.load('/kaggle/input/bekar-model/model_stage1 (2).pth', map_location=device)['model'])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:17.700309Z","iopub.execute_input":"2024-04-24T13:47:17.700667Z","iopub.status.idle":"2024-04-24T13:47:29.512598Z","shell.execute_reply.started":"2024-04-24T13:47:17.700633Z","shell.execute_reply":"2024-04-24T13:47:29.511670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:29.513724Z","iopub.execute_input":"2024-04-24T13:47:29.514032Z","iopub.status.idle":"2024-04-24T13:47:29.518878Z","shell.execute_reply.started":"2024-04-24T13:47:29.514008Z","shell.execute_reply":"2024-04-24T13:47:29.517952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sorted_vocab_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:29.520124Z","iopub.execute_input":"2024-04-24T13:47:29.520464Z","iopub.status.idle":"2024-04-24T13:47:29.537642Z","shell.execute_reply.started":"2024-04-24T13:47:29.520434Z","shell.execute_reply":"2024-04-24T13:47:29.536773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sorted_vocab_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:29.538770Z","iopub.execute_input":"2024-04-24T13:47:29.539106Z","iopub.status.idle":"2024-04-24T13:47:29.547668Z","shell.execute_reply.started":"2024-04-24T13:47:29.539076Z","shell.execute_reply":"2024-04-24T13:47:29.546672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(\"/kaggle/input/language-model-cd-v1/5gram_correct.arpa\"),\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:29.552327Z","iopub.execute_input":"2024-04-24T13:47:29.552600Z","iopub.status.idle":"2024-04-24T13:47:33.775268Z","shell.execute_reply.started":"2024-04-24T13:47:29.552578Z","shell.execute_reply":"2024-04-24T13:47:33.774470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.776260Z","iopub.execute_input":"2024-04-24T13:47:33.776518Z","iopub.status.idle":"2024-04-24T13:47:33.781059Z","shell.execute_reply.started":"2024-04-24T13:47:33.776496Z","shell.execute_reply":"2024-04-24T13:47:33.780145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr)[0]\n        return w","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.782353Z","iopub.execute_input":"2024-04-24T13:47:33.782925Z","iopub.status.idle":"2024-04-24T13:47:33.796168Z","shell.execute_reply.started":"2024-04-24T13:47:33.782894Z","shell.execute_reply":"2024-04-24T13:47:33.795274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/test-data-final/test_data_with_duration_final_v1_1703.csv\"\ntest = pd.read_csv(path)  #.head(5)\ntest.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.797343Z","iopub.execute_input":"2024-04-24T13:47:33.797770Z","iopub.status.idle":"2024-04-24T13:47:33.842614Z","shell.execute_reply.started":"2024-04-24T13:47:33.797717Z","shell.execute_reply":"2024-04-24T13:47:33.841718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['original_index'] = test['index']","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.843655Z","iopub.execute_input":"2024-04-24T13:47:33.843935Z","iopub.status.idle":"2024-04-24T13:47:33.849114Z","shell.execute_reply.started":"2024-04-24T13:47:33.843912Z","shell.execute_reply":"2024-04-24T13:47:33.848127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.850303Z","iopub.execute_input":"2024-04-24T13:47:33.850647Z","iopub.status.idle":"2024-04-24T13:47:33.860748Z","shell.execute_reply.started":"2024-04-24T13:47:33.850614Z","shell.execute_reply":"2024-04-24T13:47:33.859912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.861874Z","iopub.execute_input":"2024-04-24T13:47:33.862210Z","iopub.status.idle":"2024-04-24T13:47:33.875591Z","shell.execute_reply.started":"2024-04-24T13:47:33.862179Z","shell.execute_reply":"2024-04-24T13:47:33.874708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = list(test['path'])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.876917Z","iopub.execute_input":"2024-04-24T13:47:33.877668Z","iopub.status.idle":"2024-04-24T13:47:33.884637Z","shell.execute_reply.started":"2024-04-24T13:47:33.877636Z","shell.execute_reply":"2024-04-24T13:47:33.883763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths[:3]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.885781Z","iopub.execute_input":"2024-04-24T13:47:33.886273Z","iopub.status.idle":"2024-04-24T13:47:33.896294Z","shell.execute_reply.started":"2024-04-24T13:47:33.886249Z","shell.execute_reply":"2024-04-24T13:47:33.895427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.897292Z","iopub.execute_input":"2024-04-24T13:47:33.897554Z","iopub.status.idle":"2024-04-24T13:47:33.906040Z","shell.execute_reply.started":"2024-04-24T13:47:33.897529Z","shell.execute_reply":"2024-04-24T13:47:33.905228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import partial","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.907410Z","iopub.execute_input":"2024-04-24T13:47:33.907664Z","iopub.status.idle":"2024-04-24T13:47:33.917046Z","shell.execute_reply.started":"2024-04-24T13:47:33.907641Z","shell.execute_reply":"2024-04-24T13:47:33.916235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.918064Z","iopub.execute_input":"2024-04-24T13:47:33.918319Z","iopub.status.idle":"2024-04-24T13:47:33.927743Z","shell.execute_reply.started":"2024-04-24T13:47:33.918297Z","shell.execute_reply":"2024-04-24T13:47:33.926922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to(device)\nmodel.freeze_feature_encoder()\npred_sentence_list = []","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:33.928873Z","iopub.execute_input":"2024-04-24T13:47:33.929147Z","iopub.status.idle":"2024-04-24T13:47:34.326917Z","shell.execute_reply.started":"2024-04-24T13:47:33.929124Z","shell.execute_reply":"2024-04-24T13:47:34.325868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=1, shuffle=False,\n    num_workers=16, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:34.328130Z","iopub.execute_input":"2024-04-24T13:47:34.328429Z","iopub.status.idle":"2024-04-24T13:47:34.335078Z","shell.execute_reply.started":"2024-04-24T13:47:34.328404Z","shell.execute_reply":"2024-04-24T13:47:34.334161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\n    \nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:34.336489Z","iopub.execute_input":"2024-04-24T13:47:34.337001Z","iopub.status.idle":"2024-04-24T13:47:34.344345Z","shell.execute_reply.started":"2024-04-24T13:47:34.336968Z","shell.execute_reply":"2024-04-24T13:47:34.343352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\npred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n\n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=256*3).text\n            pred_sentence_list.append(sentence)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:47:34.345411Z","iopub.execute_input":"2024-04-24T13:47:34.345701Z","iopub.status.idle":"2024-04-24T13:48:06.714889Z","shell.execute_reply.started":"2024-04-24T13:47:34.345678Z","shell.execute_reply":"2024-04-24T13:48:06.713905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(pred_sentence_list)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.716109Z","iopub.execute_input":"2024-04-24T13:48:06.716402Z","iopub.status.idle":"2024-04-24T13:48:06.723300Z","shell.execute_reply.started":"2024-04-24T13:48:06.716377Z","shell.execute_reply":"2024-04-24T13:48:06.722319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(pred_sentence_list)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.724366Z","iopub.execute_input":"2024-04-24T13:48:06.724630Z","iopub.status.idle":"2024-04-24T13:48:06.734844Z","shell.execute_reply.started":"2024-04-24T13:48:06.724608Z","shell.execute_reply":"2024-04-24T13:48:06.733897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.735847Z","iopub.execute_input":"2024-04-24T13:48:06.736161Z","iopub.status.idle":"2024-04-24T13:48:06.746313Z","shell.execute_reply.started":"2024-04-24T13:48:06.736130Z","shell.execute_reply":"2024-04-24T13:48:06.745410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test))\ntest.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.747478Z","iopub.execute_input":"2024-04-24T13:48:06.747763Z","iopub.status.idle":"2024-04-24T13:48:06.765126Z","shell.execute_reply.started":"2024-04-24T13:48:06.747712Z","shell.execute_reply":"2024-04-24T13:48:06.764197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fun(row):\n    if len(row) == 0:\n        return \"|\"\n    return row\ntest['sentence'] = pred_sentence_list\ntest['sentence'] = test['sentence'].apply(fun)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.766120Z","iopub.execute_input":"2024-04-24T13:48:06.766378Z","iopub.status.idle":"2024-04-24T13:48:06.774513Z","shell.execute_reply.started":"2024-04-24T13:48:06.766357Z","shell.execute_reply":"2024-04-24T13:48:06.773749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.sort_values('index')","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.775703Z","iopub.execute_input":"2024-04-24T13:48:06.776103Z","iopub.status.idle":"2024-04-24T13:48:06.795342Z","shell.execute_reply.started":"2024-04-24T13:48:06.776073Z","shell.execute_reply":"2024-04-24T13:48:06.794531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.796682Z","iopub.execute_input":"2024-04-24T13:48:06.797027Z","iopub.status.idle":"2024-04-24T13:48:06.808684Z","shell.execute_reply.started":"2024-04-24T13:48:06.797001Z","shell.execute_reply":"2024-04-24T13:48:06.807858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.to_csv(\"sub1.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.809864Z","iopub.execute_input":"2024-04-24T13:48:06.810175Z","iopub.status.idle":"2024-04-24T13:48:06.821286Z","shell.execute_reply.started":"2024-04-24T13:48:06.810141Z","shell.execute_reply":"2024-04-24T13:48:06.820527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# new - Punctuation model","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/working/sub1.csv\"\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.822202Z","iopub.execute_input":"2024-04-24T13:48:06.822444Z","iopub.status.idle":"2024-04-24T13:48:06.829742Z","shell.execute_reply.started":"2024-04-24T13:48:06.822423Z","shell.execute_reply":"2024-04-24T13:48:06.828778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(path)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.830810Z","iopub.execute_input":"2024-04-24T13:48:06.831612Z","iopub.status.idle":"2024-04-24T13:48:06.842710Z","shell.execute_reply.started":"2024-04-24T13:48:06.831588Z","shell.execute_reply":"2024-04-24T13:48:06.841953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.843783Z","iopub.execute_input":"2024-04-24T13:48:06.844050Z","iopub.status.idle":"2024-04-24T13:48:06.859587Z","shell.execute_reply.started":"2024-04-24T13:48:06.844021Z","shell.execute_reply":"2024-04-24T13:48:06.858673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = list(test['sentence'])\npred_sentence_list[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.866257Z","iopub.execute_input":"2024-04-24T13:48:06.866525Z","iopub.status.idle":"2024-04-24T13:48:06.872800Z","shell.execute_reply.started":"2024-04-24T13:48:06.866503Z","shell.execute_reply":"2024-04-24T13:48:06.871956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(pred_sentence_list)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.873850Z","iopub.execute_input":"2024-04-24T13:48:06.874107Z","iopub.status.idle":"2024-04-24T13:48:06.884992Z","shell.execute_reply.started":"2024-04-24T13:48:06.874085Z","shell.execute_reply":"2024-04-24T13:48:06.884075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        sentence = \" \".join(re.sub('[\\,\\।\\?\\!\\-]', \" \", sentence).split())\n        if len(sentence) == 0:\n            sentence = \"।\"\n    except:\n        sentence = \"।\"\n    return sentence\n\npp_pred_sentence_list = [\n    postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.886283Z","iopub.execute_input":"2024-04-24T13:48:06.886935Z","iopub.status.idle":"2024-04-24T13:48:06.920900Z","shell.execute_reply.started":"2024-04-24T13:48:06.886904Z","shell.execute_reply":"2024-04-24T13:48:06.920080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.922055Z","iopub.execute_input":"2024-04-24T13:48:06.922309Z","iopub.status.idle":"2024-04-24T13:48:06.927656Z","shell.execute_reply.started":"2024-04-24T13:48:06.922287Z","shell.execute_reply":"2024-04-24T13:48:06.926894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"sentence\"] = pp_pred_sentence_list","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.928845Z","iopub.execute_input":"2024-04-24T13:48:06.929148Z","iopub.status.idle":"2024-04-24T13:48:06.938284Z","shell.execute_reply.started":"2024-04-24T13:48:06.929124Z","shell.execute_reply":"2024-04-24T13:48:06.937388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.939279Z","iopub.execute_input":"2024-04-24T13:48:06.939595Z","iopub.status.idle":"2024-04-24T13:48:06.956994Z","shell.execute_reply.started":"2024-04-24T13:48:06.939565Z","shell.execute_reply":"2024-04-24T13:48:06.956040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"pre_submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.958129Z","iopub.execute_input":"2024-04-24T13:48:06.958760Z","iopub.status.idle":"2024-04-24T13:48:06.966867Z","shell.execute_reply.started":"2024-04-24T13:48:06.958705Z","shell.execute_reply":"2024-04-24T13:48:06.966155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Punctuation","metadata":{}},{"cell_type":"code","source":"import os\nos.system('python -m pip install --no-index --find-links=../input/vakyansh-models-punctuation-models-bengali indic-nlp-library')\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForTokenClassification\nimport numpy as np\nimport pandas as pd\nimport json\nimport torch.nn as nn\nfrom indicnlp.tokenize import indic_tokenize\n\nCHECKPOINT_PATH = '/kaggle/input/punctuation-models/xlm-roberta-large_exp016_cp175.pt'\nLABEL_ENCODER_PATH = '/kaggle/input/punctuation-models/label_encoder.json'\n\nlabel_encoder_path = LABEL_ENCODER_PATH\npunctuation_dict = {'qm': '? ', 'comma': ', ', 'end': '। ', 'blank': ' ', 'hyp': '-', 'PAD': ' '}\n\nwith open(label_encoder_path) as label_encoder:\n    train_encoder = json.load(label_encoder)\n\ntokenizer = AutoTokenizer.from_pretrained(\n    '/kaggle/input/xlm-roberta-large',\n)\nmodel = AutoModelForTokenClassification.from_pretrained(\n    '/kaggle/input/xlm-roberta-large',\n    num_labels=len(train_encoder),\n    output_attentions=False,\n    output_hidden_states=False,\n)\n\ncheckpoint = torch.load(CHECKPOINT_PATH)\nmodel.load_state_dict(checkpoint['state_dict'], strict=False)\n\nmodel.eval()\nmodel.cuda()\n\n# Added model\nCHECKPOINT_PATH_1 = '/kaggle/input/punctuation-models/xlm-roberta-base_exp017_cp200.pt'\ntokenizer_1 = AutoTokenizer.from_pretrained(\n    '/kaggle/input/bengaliai-xlm-roberta-base',\n)\nmodel_1 = AutoModelForTokenClassification.from_pretrained(\n    '/kaggle/input/bengaliai-xlm-roberta-base',\n    num_labels=len(train_encoder),\n    output_attentions=False,\n    output_hidden_states=False,\n)\ncheckpoint_1 = torch.load(CHECKPOINT_PATH_1)\nmodel_1.load_state_dict(checkpoint_1['state_dict'], strict=False)\nmodel_1.eval()\nmodel_1.cuda()\n\ndef get_tokens_and_labels_indices_from_text(text):\n    # MODEL 0\n    tokenized_sentence = tokenizer.encode(text)\n    input_ids = torch.tensor([tokenized_sentence]).cuda()\n    with torch.no_grad():\n        output = model(input_ids)\n    label_indices = output[0].to('cpu').numpy()\n    tokens = tokenizer.convert_ids_to_tokens(input_ids.to('cpu').numpy()[0])\n    # MODEL 1\n    tokenized_sentence_1 = tokenizer_1.encode(text)\n    input_ids_1 = torch.tensor([tokenized_sentence_1]).cuda()\n    with torch.no_grad():\n        output_1 = model_1(input_ids_1)\n    label_indices_1 = output_1[0].to('cpu').numpy()\n    tokens_1 = tokenizer_1.convert_ids_to_tokens(input_ids_1.to('cpu').numpy()[0])\n    return tokens, label_indices, tokens_1, label_indices_1\n\ndef map_tokens_and_labels_to_word_and_punctuations(text):\n    if len(text) > 2048:\n        full_text = \" \".join(text.split())\n        if full_text[-1] not in [\"।\", \"!\", \"?\", \"-\"]:\n            full_text += \"।\"\n        return full_text\n    tokens, label_indices, tokens_1, label_indices_1 = get_tokens_and_labels_indices_from_text(text)\n    new_tokens = []\n    new_labels = []\n    for i in range(1, len(tokens) - 1):\n        if tokens[i].startswith(\"▁\"):\n            current_word = tokens[i][1:]\n            new_labels.append(label_indices[0][i])\n            for j in range(i + 1, len(tokens) - 1):\n                if not tokens[j].startswith(\"▁\"):\n                    current_word = current_word + tokens[j]\n                if tokens[j].startswith(\"▁\"):\n                    break\n            new_tokens.append(current_word)\n    full_text = ''\n    tokenized_text = indic_tokenize.trivial_tokenize_indic(text)\n    \n    if len(tokenized_text) == len(new_labels):\n        full_text_tokens = tokenized_text\n    else:\n        full_text_tokens = new_tokens\n    new_tokens_1 = []\n    new_labels_1 = []\n    for i in range(1, len(tokens_1) - 1):\n        if tokens_1[i].startswith(\"▁\"):\n            current_word_1 = tokens_1[i][1:]\n            new_labels_1.append(label_indices_1[0][i])\n            for j in range(i + 1, len(tokens_1) - 1):\n                if not tokens_1[j].startswith(\"▁\"):\n                    current_word_1 = current_word_1 + tokens_1[j]\n                if tokens_1[j].startswith(\"▁\"):\n                    break\n            new_tokens_1.append(current_word_1)\n    if len(new_labels) == len(new_labels_1):\n        for word, punctuation_0, punctuation_1 in zip(full_text_tokens, new_labels, new_labels_1):\n            x_0 = np.exp(punctuation_0 - np.max(punctuation_0))\n            pred_0 = x_0 / np.sum(x_0)\n            x_1 = np.exp(punctuation_1 - np.max(punctuation_1))\n            pred_1 = x_1 / np.sum(x_1)\n            punctuation = pred_0 * 0.95 + pred_1 * 0.05\n            punctuation = np.argmax(punctuation)\n            punctuation = list(train_encoder.keys())[list(train_encoder.values()).index(punctuation)]\n            full_text = full_text + word + punctuation_dict[punctuation]\n    else:\n        for word, punctuation in zip(full_text_tokens, new_labels):\n            punctuation = np.argmax(punctuation)\n            punctuation = list(train_encoder.keys())[list(train_encoder.values()).index(punctuation)]\n            full_text = full_text + word + punctuation_dict[punctuation]\n    \n    full_text = \" \".join(full_text.split())\n    if len(full_text) == 0:\n        full_text = \" \".join(text.split())\n        if len(full_text) == 0:\n            full_text = \"।\"\n    if full_text[-1] in [\",\", \"-\"]:\n        full_text = full_text[:-1] + \"।\"\n    if full_text[-1] not in [\"।\", \"!\", \"?\"]:\n        full_text += \"।\"\n    \n    return full_text\n\nsub = pd.read_csv(\"pre_submission.csv\")\nsub['sentence'] = sub['sentence'].apply(map_tokens_and_labels_to_word_and_punctuations)\n\nfrom bnunicodenormalizer import Normalizer\nbnorm = Normalizer()\ndef normalize(sentence):\n    word = [bnorm(word)['normalized'] for word in sentence.split()]\n    return \" \".join([w for w in word if w is not None])\nsub['sentence'] = sub['sentence'].apply(lambda x: normalize(x))","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:48:06.968122Z","iopub.execute_input":"2024-04-24T13:48:06.968367Z","iopub.status.idle":"2024-04-24T13:51:30.996798Z","shell.execute_reply.started":"2024-04-24T13:48:06.968336Z","shell.execute_reply":"2024-04-24T13:51:30.995966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:51:30.997958Z","iopub.execute_input":"2024-04-24T13:51:30.998264Z","iopub.status.idle":"2024-04-24T13:51:31.010530Z","shell.execute_reply.started":"2024-04-24T13:51:30.998237Z","shell.execute_reply":"2024-04-24T13:51:31.009622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sub[['id', 'sentence']]\nsub.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:51:31.011557Z","iopub.execute_input":"2024-04-24T13:51:31.011844Z","iopub.status.idle":"2024-04-24T13:51:31.027719Z","shell.execute_reply.started":"2024-04-24T13:51:31.011820Z","shell.execute_reply":"2024-04-24T13:51:31.026858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:51:31.028730Z","iopub.execute_input":"2024-04-24T13:51:31.029025Z","iopub.status.idle":"2024-04-24T13:51:31.042819Z","shell.execute_reply.started":"2024-04-24T13:51:31.029002Z","shell.execute_reply":"2024-04-24T13:51:31.042002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T13:51:31.043989Z","iopub.execute_input":"2024-04-24T13:51:31.044312Z","iopub.status.idle":"2024-04-24T13:51:31.053470Z","shell.execute_reply.started":"2024-04-24T13:51:31.044283Z","shell.execute_reply":"2024-04-24T13:51:31.052692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}