{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:18:08.194078Z","iopub.execute_input":"2023-07-24T16:18:08.194449Z","iopub.status.idle":"2023-07-24T16:18:09.222687Z","shell.execute_reply.started":"2023-07-24T16:18:08.194418Z","shell.execute_reply":"2023-07-24T16:18:09.221085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade pip","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:18:10.675376Z","iopub.execute_input":"2023-07-24T16:18:10.675755Z","iopub.status.idle":"2023-07-24T16:18:22.592339Z","shell.execute_reply.started":"2023-07-24T16:18:10.675719Z","shell.execute_reply":"2023-07-24T16:18:22.590983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:18:22.596074Z","iopub.execute_input":"2023-07-24T16:18:22.596403Z","iopub.status.idle":"2023-07-24T16:19:35.325031Z","shell.execute_reply.started":"2023-07-24T16:18:22.596373Z","shell.execute_reply":"2023-07-24T16:19:35.323861Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade numpy","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:21:08.886777Z","iopub.execute_input":"2023-07-24T16:21:08.887662Z","iopub.status.idle":"2023-07-24T16:21:28.226404Z","shell.execute_reply.started":"2023-07-24T16:21:08.887628Z","shell.execute_reply":"2023-07-24T16:21:28.224894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install numpy==1.23.0","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:21:37.606974Z","iopub.execute_input":"2023-07-24T16:21:37.607383Z","iopub.status.idle":"2023-07-24T16:21:53.359007Z","shell.execute_reply.started":"2023-07-24T16:21:37.607349Z","shell.execute_reply":"2023-07-24T16:21:53.357528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --force-reinstall tensorflow-io","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:21:58.509715Z","iopub.execute_input":"2023-07-24T16:21:58.510888Z","iopub.status.idle":"2023-07-24T16:22:13.005904Z","shell.execute_reply.started":"2023-07-24T16:21:58.510832Z","shell.execute_reply":"2023-07-24T16:22:13.004788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport pandas as pd\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:22:18.201444Z","iopub.execute_input":"2023-07-24T16:22:18.201897Z","iopub.status.idle":"2023-07-24T16:22:18.211112Z","shell.execute_reply.started":"2023-07-24T16:22:18.201843Z","shell.execute_reply":"2023-07-24T16:22:18.210105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 16\nTEST_DIRECTORY = '/kaggle/input/bengaliai-speech/test_mp3s'\npaths = glob(os.path.join(TEST_DIRECTORY,'*.mp3'))\nprint(paths[:2])","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:22:30.156778Z","iopub.execute_input":"2023-07-24T16:22:30.157848Z","iopub.status.idle":"2023-07-24T16:22:30.174323Z","shell.execute_reply.started":"2023-07-24T16:22:30.157784Z","shell.execute_reply":"2023-07-24T16:22:30.173226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    my_model_name = '../input/yellowking-dlsprint-model/YellowKing_model'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:22:41.397871Z","iopub.execute_input":"2023-07-24T16:22:41.398247Z","iopub.status.idle":"2023-07-24T16:22:41.403298Z","shell.execute_reply.started":"2023-07-24T16:22:41.398218Z","shell.execute_reply":"2023-07-24T16:22:41.402296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:22:52.542759Z","iopub.execute_input":"2023-07-24T16:22:52.543174Z","iopub.status.idle":"2023-07-24T16:24:31.562126Z","shell.execute_reply.started":"2023-07-24T16:22:52.543141Z","shell.execute_reply":"2023-07-24T16:24:31.560998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:24:31.564145Z","iopub.execute_input":"2023-07-24T16:24:31.564753Z","iopub.status.idle":"2023-07-24T16:24:52.916982Z","shell.execute_reply.started":"2023-07-24T16:24:31.564717Z","shell.execute_reply":"2023-07-24T16:24:52.915964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"speech, sr = librosa.load('/kaggle/input/bengaliai-speech/test_mp3s/0f3dac00655e.mp3', sr=processor.feature_extractor.sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:24:52.918471Z","iopub.execute_input":"2023-07-24T16:24:52.919101Z","iopub.status.idle":"2023-07-24T16:25:02.595544Z","shell.execute_reply.started":"2023-07-24T16:24:52.919060Z","shell.execute_reply":"2023-07-24T16:25:02.594437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM([speech]*2, chunk_length_s=112, stride_length_s=None)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:25:02.597942Z","iopub.execute_input":"2023-07-24T16:25:02.599075Z","iopub.status.idle":"2023-07-24T16:25:08.940530Z","shell.execute_reply.started":"2023-07-24T16:25:02.599030Z","shell.execute_reply":"2023-07-24T16:25:08.939144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:25:20.531772Z","iopub.execute_input":"2023-07-24T16:25:20.532200Z","iopub.status.idle":"2023-07-24T16:25:20.540309Z","shell.execute_reply.started":"2023-07-24T16:25:20.532166Z","shell.execute_reply":"2023-07-24T16:25:20.539237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioDataset(Dataset):\n    def __init__(self, paths):\n        self.paths = paths\n    def __len__(self):\n        return len(self.paths)\n    def __getitem__(self,idx):\n        speech, sr = librosa.load(self.paths[idx], sr=processor.feature_extractor.sampling_rate) \n#         print(speech.shape)\n        return speech","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:25:31.846151Z","iopub.execute_input":"2023-07-24T16:25:31.846553Z","iopub.status.idle":"2023-07-24T16:25:31.853561Z","shell.execute_reply.started":"2023-07-24T16:25:31.846521Z","shell.execute_reply":"2023-07-24T16:25:31.851871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = AudioDataset(paths)\ndataset[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:27.374695Z","iopub.execute_input":"2023-07-24T16:28:27.375961Z","iopub.status.idle":"2023-07-24T16:28:27.397045Z","shell.execute_reply.started":"2023-07-24T16:28:27.375896Z","shell.execute_reply":"2023-07-24T16:28:27.396030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda:0'","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:29.621357Z","iopub.execute_input":"2023-07-24T16:28:29.621726Z","iopub.status.idle":"2023-07-24T16:28:29.629386Z","shell.execute_reply.started":"2023-07-24T16:28:29.621694Z","shell.execute_reply":"2023-07-24T16:28:29.628387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn_padd(batch):\n    '''\n    Padds batch of variable length\n\n    note: it converts things ToTensor manually here since the ToTensor transform\n    assume it takes in images rather than arbitrary tensors.\n    '''\n    ## get sequence lengths\n    lengths = torch.tensor([ t.shape[0] for t in batch ])\n    ## padd\n    batch = [ torch.Tensor(t) for t in batch ]\n    batch = torch.nn.utils.rnn.pad_sequence(batch)\n    ## compute mask\n    mask = (batch != 0)\n    return batch, lengths, mask","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:31.116673Z","iopub.execute_input":"2023-07-24T16:28:31.117395Z","iopub.status.idle":"2023-07-24T16:28:31.123741Z","shell.execute_reply.started":"2023-07-24T16:28:31.117358Z","shell.execute_reply":"2023-07-24T16:28:31.122537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataloader = DataLoader(dataset, batch_size=32, shuffle=False, num_workers=8, collate_fn=collate_fn_padd)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:33.245623Z","iopub.execute_input":"2023-07-24T16:28:33.246062Z","iopub.status.idle":"2023-07-24T16:28:33.252208Z","shell.execute_reply.started":"2023-07-24T16:28:33.246028Z","shell.execute_reply":"2023-07-24T16:28:33.250828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_all = []\nfor batch, lengths, mask in dataloader:\n    preds = my_asrLM(list(batch.numpy().transpose()))\n    preds_all+=preds","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:34.791423Z","iopub.execute_input":"2023-07-24T16:28:34.791815Z","iopub.status.idle":"2023-07-24T16:28:36.317905Z","shell.execute_reply.started":"2023-07-24T16:28:34.791766Z","shell.execute_reply":"2023-07-24T16:28:36.316575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer \n\n\nbnorm = Normalizer()\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:28:55.578192Z","iopub.execute_input":"2023-07-24T16:28:55.578649Z","iopub.status.idle":"2023-07-24T16:28:55.591094Z","shell.execute_reply.started":"2023-07-24T16:28:55.578611Z","shell.execute_reply":"2023-07-24T16:28:55.590027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.DataFrame(\n    {\n        \"id\":[p.split(os.sep)[-1].replace('.mp3','') for p in paths],\n        \"sentence\":[p['text']for p in preds_all]\n    }\n)\ndf.sentence= df.sentence.apply(lambda x:normalize(x))\ndf.sentence= df.sentence.apply(lambda x:dari(x))","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:29:07.905194Z","iopub.execute_input":"2023-07-24T16:29:07.905570Z","iopub.status.idle":"2023-07-24T16:29:07.947115Z","shell.execute_reply.started":"2023-07-24T16:29:07.905540Z","shell.execute_reply":"2023-07-24T16:29:07.946163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:29:16.778798Z","iopub.execute_input":"2023-07-24T16:29:16.779220Z","iopub.status.idle":"2023-07-24T16:29:16.796619Z","shell.execute_reply.started":"2023-07-24T16:29:16.779188Z","shell.execute_reply":"2023-07-24T16:29:16.795317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-24T16:29:27.072258Z","iopub.execute_input":"2023-07-24T16:29:27.072619Z","iopub.status.idle":"2023-07-24T16:29:27.084737Z","shell.execute_reply.started":"2023-07-24T16:29:27.072588Z","shell.execute_reply":"2023-07-24T16:29:27.083646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}