{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Inference code for YellowKing's model from  DL Sprint 2022\nhttps://www.kaggle.com/code/sameen53/yellowking-dlsprint-inference","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:16:43.997077Z","iopub.execute_input":"2023-08-29T05:16:43.997481Z","iopub.status.idle":"2023-08-29T05:16:45.012494Z","shell.execute_reply.started":"2023-08-29T05:16:43.997439Z","shell.execute_reply":"2023-08-29T05:16:45.011091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:16:45.014985Z","iopub.execute_input":"2023-08-29T05:16:45.016007Z","iopub.status.idle":"2023-08-29T05:18:21.751660Z","shell.execute_reply.started":"2023-08-29T05:16:45.015957Z","shell.execute_reply":"2023-08-29T05:18:21.750067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport pandas as pd\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:18:21.755337Z","iopub.execute_input":"2023-08-29T05:18:21.755825Z","iopub.status.idle":"2023-08-29T05:18:32.466818Z","shell.execute_reply.started":"2023-08-29T05:18:21.755778Z","shell.execute_reply":"2023-08-29T05:18:32.465645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 16\nTEST_DIRECTORY = '/kaggle/input/bengaliai-speech/test_mp3s'\npaths = glob(os.path.join(TEST_DIRECTORY,'*.mp3'))\nprint(paths[:2])","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:20:29.251176Z","iopub.execute_input":"2023-08-29T05:20:29.252386Z","iopub.status.idle":"2023-08-29T05:20:29.264589Z","shell.execute_reply.started":"2023-08-29T05:20:29.252343Z","shell.execute_reply":"2023-08-29T05:20:29.263400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass CFG:\n    my_model_name = '../input/yellowking-dlsprint-model/YellowKing_model'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:20:30.821669Z","iopub.execute_input":"2023-08-29T05:20:30.822972Z","iopub.status.idle":"2023-08-29T05:20:30.828358Z","shell.execute_reply.started":"2023-08-29T05:20:30.822914Z","shell.execute_reply":"2023-08-29T05:20:30.827049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:20:49.149526Z","iopub.execute_input":"2023-08-29T05:20:49.149948Z","iopub.status.idle":"2023-08-29T05:22:26.451857Z","shell.execute_reply.started":"2023-08-29T05:20:49.149910Z","shell.execute_reply":"2023-08-29T05:22:26.450757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)\nmy_asrLM = pipeline(\n    \"automatic-speech-recognition\",\n    model=CFG.my_model_name,\n    feature_extractor =processor.feature_extractor,\n    tokenizer= processor.tokenizer,\n    decoder=processor.decoder,\n    device=0\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:27:08.456882Z","iopub.execute_input":"2023-08-29T05:27:08.457708Z","iopub.status.idle":"2023-08-29T05:27:24.199348Z","shell.execute_reply.started":"2023-08-29T05:27:08.457667Z","shell.execute_reply":"2023-08-29T05:27:24.198172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"speech, sr = librosa.load('/kaggle/input/bengaliai-speech/test_mp3s/0f3dac00655e.mp3', sr=processor.feature_extractor.sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:27:44.898974Z","iopub.execute_input":"2023-08-29T05:27:44.899371Z","iopub.status.idle":"2023-08-29T05:27:48.211892Z","shell.execute_reply.started":"2023-08-29T05:27:44.899336Z","shell.execute_reply":"2023-08-29T05:27:48.210533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my_asrLM([speech]*2, chunk_length_s=112, stride_length_s=None)\nmy_asrLM(speech, chunk_length_s=112, stride_length_s=None)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:30:15.250111Z","iopub.execute_input":"2023-08-29T05:30:15.251321Z","iopub.status.idle":"2023-08-29T05:30:15.310638Z","shell.execute_reply.started":"2023-08-29T05:30:15.251280Z","shell.execute_reply":"2023-08-29T05:30:15.309580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:30:20.716504Z","iopub.execute_input":"2023-08-29T05:30:20.716934Z","iopub.status.idle":"2023-08-29T05:30:20.725052Z","shell.execute_reply.started":"2023-08-29T05:30:20.716897Z","shell.execute_reply":"2023-08-29T05:30:20.723775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Following Sample Submission:**","metadata":{}},{"cell_type":"code","source":"class AudioDataset(Dataset):\n    def __init__(self, paths):\n        self.paths = paths\n    def __len__(self):\n        return len(self.paths)\n    def __getitem__(self,idx):\n        speech, sr = librosa.load(self.paths[idx], sr=processor.feature_extractor.sampling_rate) \n#         print(speech.shape)\n        return speech","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:30:28.579318Z","iopub.execute_input":"2023-08-29T05:30:28.579740Z","iopub.status.idle":"2023-08-29T05:30:28.587453Z","shell.execute_reply.started":"2023-08-29T05:30:28.579685Z","shell.execute_reply":"2023-08-29T05:30:28.586458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = AudioDataset(paths)\ndataset[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:30:35.406788Z","iopub.execute_input":"2023-08-29T05:30:35.407539Z","iopub.status.idle":"2023-08-29T05:30:36.674053Z","shell.execute_reply.started":"2023-08-29T05:30:35.407499Z","shell.execute_reply":"2023-08-29T05:30:36.672758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda:0'","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:30:40.362442Z","iopub.execute_input":"2023-08-29T05:30:40.362903Z","iopub.status.idle":"2023-08-29T05:30:40.368516Z","shell.execute_reply.started":"2023-08-29T05:30:40.362861Z","shell.execute_reply":"2023-08-29T05:30:40.367174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn_padd(batch):\n    '''\n    Padds batch of variable length\n\n    note: it converts things ToTensor manually here since the ToTensor transform\n    assume it takes in images rather than arbitrary tensors.\n    '''\n    ## get sequence lengths\n    lengths = torch.tensor([ t.shape[0] for t in batch ])\n    ## padd\n    batch = [ torch.Tensor(t) for t in batch ]\n    batch = torch.nn.utils.rnn.pad_sequence(batch)\n    ## compute mask\n    mask = (batch != 0)\n    return batch, lengths, mask\n# def collate_fn_padd(batch):\n#     '''\n#     Pads batch of variable length\n\n#     Note: It converts things ToTensor manually here since the ToTensor transform\n#     assumes it takes in images rather than arbitrary tensors.\n#     '''\n#     ## get sequence lengths\n#     lengths = torch.tensor([t.shape[0] for t in batch])\n\n#     ## convert batch to tensor\n#     batch = torch.stack([torch.Tensor(t) for t in batch])\n\n#     ## compute mask\n#     mask = (batch != 0)\n\n#     return batch, lengths, mask\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:33:02.145067Z","iopub.execute_input":"2023-08-29T05:33:02.146140Z","iopub.status.idle":"2023-08-29T05:33:02.154976Z","shell.execute_reply.started":"2023-08-29T05:33:02.146095Z","shell.execute_reply":"2023-08-29T05:33:02.153802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataloader = DataLoader(dataset, batch_size=32, shuffle=False, num_workers=8, collate_fn=collate_fn_padd)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:33:08.704456Z","iopub.execute_input":"2023-08-29T05:33:08.704883Z","iopub.status.idle":"2023-08-29T05:33:08.710671Z","shell.execute_reply.started":"2023-08-29T05:33:08.704845Z","shell.execute_reply":"2023-08-29T05:33:08.709486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_all = []\nfor batch, lengths, mask in dataloader:\n    preds = my_asrLM(list(batch.numpy().transpose()))\n    preds_all+=preds","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:33:12.374467Z","iopub.execute_input":"2023-08-29T05:33:12.374906Z","iopub.status.idle":"2023-08-29T05:33:18.900136Z","shell.execute_reply.started":"2023-08-29T05:33:12.374869Z","shell.execute_reply":"2023-08-29T05:33:18.898847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from bnunicodenormalizer import Normalizer \n\n\n# bnorm = Normalizer()\n# def normalize(sen):\n#     _words = [bnorm(word)['normalized']  for word in sen.split()]\n#     return \" \".join([word for word in _words if word is not None])\n\n# def dari(sentence):\n#     try:\n#         if sentence[-1]!=\"।\":\n#             sentence+=\"।\"\n#     except:\n#         print(sentence)\n#     return sentence\nfrom bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\n\ndef normalize(sen):\n    _words = [bnorm(word)['normalized'] for word in sen.split() if bnorm(word) is not None]\n    return \" \".join(_words)\n\ndef dari(sentence):\n    if not sentence.endswith(\"।\"):\n        sentence += \"।\"\n    return sentence\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:34:52.364870Z","iopub.execute_input":"2023-08-29T05:34:52.365283Z","iopub.status.idle":"2023-08-29T05:34:52.372700Z","shell.execute_reply.started":"2023-08-29T05:34:52.365248Z","shell.execute_reply":"2023-08-29T05:34:52.371735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df= pd.DataFrame(\n#     {\n#         \"id\":[p.split(os.sep)[-1].replace('.mp3','') for p in paths],\n#         \"sentence\":[p['text']for p in preds_all]\n#     }\n# )\n# df.sentence= df.sentence.apply(lambda x:normalize(x))\n# df.sentence= df.sentence.apply(lambda x:dari(x))\ndf = pd.DataFrame({\n    \"id\": [p.split(os.sep)[-1].replace('.mp3', '') for p in paths],\n    \"sentence\": [dari(normalize(p['text'])) for p in preds_all]\n})\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:35:46.007695Z","iopub.execute_input":"2023-08-29T05:35:46.008470Z","iopub.status.idle":"2023-08-29T05:35:46.044486Z","shell.execute_reply.started":"2023-08-29T05:35:46.008431Z","shell.execute_reply":"2023-08-29T05:35:46.043486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:35:49.408282Z","iopub.execute_input":"2023-08-29T05:35:49.408755Z","iopub.status.idle":"2023-08-29T05:35:49.427163Z","shell.execute_reply.started":"2023-08-29T05:35:49.408687Z","shell.execute_reply":"2023-08-29T05:35:49.426069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-29T05:35:55.213435Z","iopub.execute_input":"2023-08-29T05:35:55.213852Z","iopub.status.idle":"2023-08-29T05:35:55.223864Z","shell.execute_reply.started":"2023-08-29T05:35:55.213814Z","shell.execute_reply":"2023-08-29T05:35:55.222764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}