{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install --no-index --no-deps /kaggle/input/4-34-0/transformers-4.34.0-py3-none-any.whl\n# !pip install --no-index --no-deps /kaggle/input/tokenizers-0-14-1/tokenizers-0.14.1-cp310-none-win_amd64.whl\n!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:11:59.743462Z","iopub.execute_input":"2023-10-17T20:11:59.743887Z","iopub.status.idle":"2023-10-17T20:13:01.118348Z","shell.execute_reply.started":"2023-10-17T20:11:59.743863Z","shell.execute_reply":"2023-10-17T20:13:01.117425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall torch -y\n!pip install --no-index --no-deps /kaggle/input/torch-2-1-0/torch-2.1.0cu118-cp310-cp310-linux_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:13:01.120112Z","iopub.execute_input":"2023-10-17T20:13:01.120354Z","iopub.status.idle":"2023-10-17T20:14:16.469485Z","shell.execute_reply.started":"2023-10-17T20:13:01.120332Z","shell.execute_reply":"2023-10-17T20:14:16.468615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport torch\nimport json\nimport librosa\nfrom transformers import Wav2Vec2ForCTC, Wav2Vec2CTCTokenizer, Wav2Vec2FeatureExtractor, Wav2Vec2Processor\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.nn.utils.rnn import pad_sequence\nfrom tqdm import tqdm\nfrom bnunicodenormalizer import Normalizer","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:16.470739Z","iopub.execute_input":"2023-10-17T20:14:16.470992Z","iopub.status.idle":"2023-10-17T20:14:28.080702Z","shell.execute_reply.started":"2023-10-17T20:14:16.470968Z","shell.execute_reply":"2023-10-17T20:14:28.080051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BengaliDataset(Dataset):\n    def __init__(self, df, processor):\n        self.df = df\n        self.processor = processor\n\n    def __getitem__(self, idx):\n        audio_path = self.df.loc[idx]['path']\n        audio_array = self.read_audio(audio_path)\n\n        inputs = self.processor(\n            audio_array,\n            sampling_rate=16_000,\n            return_tensors='pt'\n        )\n\n        with self.processor.as_target_processor():\n            labels = self.processor(self.df.loc[idx]['sentence']).input_ids\n\n        return {'input_values': inputs['input_values'].squeeze(0), 'labels': labels}\n\n    def __len__(self):\n        return len(self.df)\n\n    def read_audio(self, mp3_path):\n        target_sr = 16000  # Set the target sampling rate\n\n        audio, sr = librosa.load(mp3_path, sr=None)  # Load with original sampling rate\n        audio_array = librosa.resample(audio, orig_sr=sr, target_sr=target_sr)\n\n        return audio_array","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:28.082358Z","iopub.execute_input":"2023-10-17T20:14:28.082574Z","iopub.status.idle":"2023-10-17T20:14:28.088423Z","shell.execute_reply.started":"2023-10-17T20:14:28.082555Z","shell.execute_reply":"2023-10-17T20:14:28.087843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(\"/kaggle/input/checkpoint25k-fleur\")","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:28.089082Z","iopub.execute_input":"2023-10-17T20:14:28.089584Z","iopub.status.idle":"2023-10-17T20:14:38.408164Z","shell.execute_reply.started":"2023-10-17T20:14:28.089563Z","shell.execute_reply":"2023-10-17T20:14:38.407362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.409348Z","iopub.execute_input":"2023-10-17T20:14:38.409579Z","iopub.status.idle":"2023-10-17T20:14:38.416547Z","shell.execute_reply.started":"2023-10-17T20:14:38.409559Z","shell.execute_reply":"2023-10-17T20:14:38.416033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor = Wav2Vec2Processor.from_pretrained(\"/kaggle/input/checkpoint25k-fleur\")","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.417260Z","iopub.execute_input":"2023-10-17T20:14:38.417693Z","iopub.status.idle":"2023-10-17T20:14:38.447847Z","shell.execute_reply.started":"2023-10-17T20:14:38.417671Z","shell.execute_reply":"2023-10-17T20:14:38.447283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/bengaliai-speech/sample_submission.csv\", dtype={\"id\": str})\n\ntest['path'] = test['id'].apply(lambda x: os.path.join('/kaggle/input/bengaliai-speech/test_mp3s', x+'.mp3'))\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.448840Z","iopub.execute_input":"2023-10-17T20:14:38.449073Z","iopub.status.idle":"2023-10-17T20:14:38.467865Z","shell.execute_reply.started":"2023-10-17T20:14:38.449054Z","shell.execute_reply":"2023-10-17T20:14:38.467149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 10\ntest_ds = BengaliDataset(test, processor)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.469037Z","iopub.execute_input":"2023-10-17T20:14:38.469291Z","iopub.status.idle":"2023-10-17T20:14:38.474666Z","shell.execute_reply.started":"2023-10-17T20:14:38.469269Z","shell.execute_reply":"2023-10-17T20:14:38.474055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nwith open(\"/kaggle/input/checkpoint25k-fleur/vocab.json\") as fopen:\n    vocab = json.load(fopen)\n    vocab = {v: k for k, v in vocab.items()}\n\nvocab = [vocab[i] for i in range(len(vocab))] \nvocab","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.479172Z","iopub.execute_input":"2023-10-17T20:14:38.479418Z","iopub.status.idle":"2023-10-17T20:14:38.486941Z","shell.execute_reply.started":"2023-10-17T20:14:38.479398Z","shell.execute_reply":"2023-10-17T20:14:38.486268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import kenlm\nfrom pyctcdecode import build_ctcdecoder\n\nlm = '/kaggle/input/newdataset-3gram-kenlm/out.arpa'\nkenlm_model = lm\ndecoder = build_ctcdecoder(\n    vocab,\n    kenlm_model,\n    alpha=0.5,\n    beta=1.0,\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:38.488099Z","iopub.execute_input":"2023-10-17T20:14:38.488335Z","iopub.status.idle":"2023-10-17T20:14:47.270987Z","shell.execute_reply.started":"2023-10-17T20:14:38.488315Z","shell.execute_reply":"2023-10-17T20:14:47.270207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_collate_fn(batch):\n    input_values = [item['input_values'] for item in batch]\n    labels = [item['labels'] for item in batch]\n\n    input_values_padded = pad_sequence(input_values, batch_first=True, padding_value=0.0)\n\n    labels = [torch.tensor(label) for label in labels]\n\n    return {'input_values': input_values_padded, 'labels': labels}","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:47.272172Z","iopub.execute_input":"2023-10-17T20:14:47.272436Z","iopub.status.idle":"2023-10-17T20:14:47.279497Z","shell.execute_reply.started":"2023-10-17T20:14:47.272401Z","shell.execute_reply":"2023-10-17T20:14:47.278623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataLoader(\n    test_ds, batch_size=batch_size, shuffle=False, num_workers=2, collate_fn=custom_collate_fn\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:47.280548Z","iopub.execute_input":"2023-10-17T20:14:47.280823Z","iopub.status.idle":"2023-10-17T20:14:47.302822Z","shell.execute_reply.started":"2023-10-17T20:14:47.280787Z","shell.execute_reply":"2023-10-17T20:14:47.302175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to('cuda')\nmodel.half()","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:47.303779Z","iopub.execute_input":"2023-10-17T20:14:47.304024Z","iopub.status.idle":"2023-10-17T20:14:51.804327Z","shell.execute_reply.started":"2023-10-17T20:14:47.303982Z","shell.execute_reply":"2023-10-17T20:14:51.803573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor.tokenizer.pad_token_id","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:51.805475Z","iopub.execute_input":"2023-10-17T20:14:51.805732Z","iopub.status.idle":"2023-10-17T20:14:51.810840Z","shell.execute_reply.started":"2023-10-17T20:14:51.805700Z","shell.execute_reply":"2023-10-17T20:14:51.810317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.__version__","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:51.811555Z","iopub.execute_input":"2023-10-17T20:14:51.812205Z","iopub.status.idle":"2023-10-17T20:14:51.823602Z","shell.execute_reply.started":"2023-10-17T20:14:51.812183Z","shell.execute_reply":"2023-10-17T20:14:51.822938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(\"cuda\", non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n\n        for l in y:\n            \n            out = decoder.decode_beams(l,prune_history = True)\n            d_lm, lm_state, timesteps, logit_score, lm_score = out[0]\n            if len(d_lm)==0:\n                d_lm = \" \"\n            sentences.append(d_lm)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:14:51.824751Z","iopub.execute_input":"2023-10-17T20:14:51.824973Z","iopub.status.idle":"2023-10-17T20:15:03.555379Z","shell.execute_reply.started":"2023-10-17T20:14:51.824954Z","shell.execute_reply":"2023-10-17T20:15:03.554565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:15:03.556534Z","iopub.execute_input":"2023-10-17T20:15:03.556814Z","iopub.status.idle":"2023-10-17T20:15:03.562354Z","shell.execute_reply.started":"2023-10-17T20:15:03.556789Z","shell.execute_reply":"2023-10-17T20:15:03.561793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_sentences = [postprocess(s) for s in tqdm(sentences)]","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:15:03.563229Z","iopub.execute_input":"2023-10-17T20:15:03.563931Z","iopub.status.idle":"2023-10-17T20:15:03.584795Z","shell.execute_reply.started":"2023-10-17T20:15:03.563907Z","shell.execute_reply":"2023-10-17T20:15:03.584072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_sentences","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:21:50.056915Z","iopub.execute_input":"2023-10-17T20:21:50.057527Z","iopub.status.idle":"2023-10-17T20:21:50.062262Z","shell.execute_reply.started":"2023-10-17T20:21:50.057500Z","shell.execute_reply":"2023-10-17T20:21:50.061516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test[['id','sentence']]\ntest.sentence = pp_sentences\ntest.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:15:03.591689Z","iopub.execute_input":"2023-10-17T20:15:03.592245Z","iopub.status.idle":"2023-10-17T20:15:03.615071Z","shell.execute_reply.started":"2023-10-17T20:15:03.592224Z","shell.execute_reply":"2023-10-17T20:15:03.614342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_file = pd.read_csv(\"submission.csv\")\ncsv_file.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-17T20:15:03.616238Z","iopub.execute_input":"2023-10-17T20:15:03.616454Z","iopub.status.idle":"2023-10-17T20:15:03.630794Z","shell.execute_reply.started":"2023-10-17T20:15:03.616436Z","shell.execute_reply":"2023-10-17T20:15:03.630034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}