{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Update: The inference will be done on .wav files","metadata":{}},{"cell_type":"code","source":"import os\nfrom glob import glob\nfrom tqdm import tqdm\nimport pandas as pd\ntqdm.pandas()\n\n# CHANGE ACCORDINGLY\nBATCH_SIZE = 16\nTEST_DIRECTORY = '/kaggle/input/dlsprint/test_files'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-18T15:19:09.564507Z","iopub.execute_input":"2022-08-18T15:19:09.564914Z","iopub.status.idle":"2022-08-18T15:19:09.571799Z","shell.execute_reply.started":"2022-08-18T15:19:09.564880Z","shell.execute_reply":"2022-08-18T15:19:09.570556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Please refactor your inference code into the below functions. It doesn't matter which function you use to ultimately infer on the test set, you can use any one. But be sure to implement working versions of these function formats.","metadata":{}},{"cell_type":"markdown","source":"# Infer on a single data","metadata":{}},{"cell_type":"code","source":"def infer(audio_path):\n  '''\n    infers on a signle audio\n    args:\n      audio_path  : the path to audio file <string>\n    returns:\n      bangla predicted text <string>\n  '''\n  # your code goes here","metadata":{"execution":{"iopub.status.busy":"2022-08-18T15:19:11.930929Z","iopub.execute_input":"2022-08-18T15:19:11.931342Z","iopub.status.idle":"2022-08-18T15:19:11.936693Z","shell.execute_reply.started":"2022-08-18T15:19:11.931308Z","shell.execute_reply":"2022-08-18T15:19:11.935693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer on a batch of data - MOST IMPORTANT","metadata":{}},{"cell_type":"code","source":"def batch_infer(audio_paths, batch_size=BATCH_SIZE):\n    '''\n    infers on a batch of audio\n    args:\n      audio_paths  : list of path to audio files <list of string>\n    returns:\n      bangla predicted texts <list of string>\n    '''\n    # your code goes her\n    return [\"\"]*len(audio_paths); # delete this line and return your predicted sentences as a list","metadata":{"execution":{"iopub.status.busy":"2022-08-18T15:21:01.398799Z","iopub.execute_input":"2022-08-18T15:21:01.399207Z","iopub.status.idle":"2022-08-18T15:21:01.405170Z","shell.execute_reply.started":"2022-08-18T15:21:01.399173Z","shell.execute_reply":"2022-08-18T15:21:01.403743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer on a directory","metadata":{}},{"cell_type":"code","source":"def directory_infer(audio_dir):\n    '''\n    infers on a directory that contains audio files\n    args:\n      audio_dir  : directory that contains some audio files <string>\n    returns:\n      a dataframe that contains 2 columns:\n        * path <string>\n        * sentence <string>\n    '''\n    # list all audio files\n\n    audio_paths=[audio_path for audio_path in tqdm(glob(os.path.join(audio_dir,\"*.*\")))]\n    sentences=[]\n    for idx in tqdm(range(0,len(audio_paths),BATCH_SIZE)):\n        batch_paths=audio_paths[idx:idx+BATCH_SIZE]\n        sentences+=batch_infer(batch_paths)\n    df=pd.DataFrame({\"path\":audio_paths,\"sentence\":sentences})\n    return df ","metadata":{"execution":{"iopub.status.busy":"2022-08-18T15:21:01.927864Z","iopub.execute_input":"2022-08-18T15:21:01.928284Z","iopub.status.idle":"2022-08-18T15:21:01.935713Z","shell.execute_reply.started":"2022-08-18T15:21:01.928249Z","shell.execute_reply":"2022-08-18T15:21:01.934623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = directory_infer(TEST_DIRECTORY)\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T15:21:02.227491Z","iopub.execute_input":"2022-08-18T15:21:02.228644Z","iopub.status.idle":"2022-08-18T15:21:02.298105Z","shell.execute_reply.started":"2022-08-18T15:21:02.228601Z","shell.execute_reply":"2022-08-18T15:21:02.297296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Your code must output a submission.csv file in the end with predictions on the test_files","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}