{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Most of the ASR models use wav files as input. The audios provided here is in mp3 format. So for convenience it is better to convert the mp3s to wavs. In this notebook, we'll try to do that. This is actually a fork of my [previous notebook](https://www.kaggle.com/code/mbmmurad/faster-way-to-convert-mp3-to-wav-using-joblib)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T04:05:50.633524Z","iopub.execute_input":"2023-07-18T04:05:50.634006Z","iopub.status.idle":"2023-07-18T04:05:51.575886Z","shell.execute_reply.started":"2023-07-18T04:05:50.633969Z","shell.execute_reply":"2023-07-18T04:05:51.574744Z"}}},{"cell_type":"code","source":"import os\nimport cv2\nimport skimage.io\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport pandas as pd\nimport numpy as np\nimport shutil\n\nfrom pydub import AudioSegment\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:11.021779Z","iopub.execute_input":"2023-07-18T10:50:11.022740Z","iopub.status.idle":"2023-07-18T10:50:12.464924Z","shell.execute_reply.started":"2023-07-18T10:50:11.022686Z","shell.execute_reply":"2023-07-18T10:50:12.463504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll need to convert the train files. but there's a problem.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/bengaliai-speech/train.csv\")\ndisplay(df.head())\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:12.467798Z","iopub.execute_input":"2023-07-18T10:50:12.468284Z","iopub.status.idle":"2023-07-18T10:50:18.661927Z","shell.execute_reply.started":"2023-07-18T10:50:12.468237Z","shell.execute_reply":"2023-07-18T10:50:18.660501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The training set has 963636 samples. converting them to wavs will take more than 12hours and probably more than 19.5GB(which is kaggle's maximum output size limit). \n\nTo tackle this, we'll need to convert these audios in different notebooks. We'll have to divide the audios into twenty folds and convert them in seperate notebooks","metadata":{}},{"cell_type":"code","source":"ROOT_PATH = \"/kaggle/input/bengaliai-speech/train_mp3s\"\nOUTPUT_DIR = \"./train_files_wav\"\nos.mkdir(OUTPUT_DIR)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:18.664092Z","iopub.execute_input":"2023-07-18T10:50:18.664583Z","iopub.status.idle":"2023-07-18T10:50:18.670412Z","shell.execute_reply.started":"2023-07-18T10:50:18.664537Z","shell.execute_reply":"2023-07-18T10:50:18.669512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_files = df['id'].apply(lambda x:x+\".mp3\").tolist()[:df.shape[0]//20]","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:18.673184Z","iopub.execute_input":"2023-07-18T10:50:18.674026Z","iopub.status.idle":"2023-07-18T10:50:19.166704Z","shell.execute_reply.started":"2023-07-18T10:50:18.673992Z","shell.execute_reply":"2023-07-18T10:50:19.165190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Function to convert a single audio file\n\ndef save_fn(filename):\n    \n    path = f\"{ROOT_PATH}/{filename}\"\n    save_path = f\"{OUTPUT_DIR}\"\n    if not os.path.exists(save_path):\n        os.makedirs(save_path, exist_ok=True)\n    \n    if os.path.exists(path):\n        try:\n            sound = AudioSegment.from_mp3(path)\n            #sound = sound.set_frame_rate(32000)\n            sound.export(f\"{save_path}/{filename[:-4]}.wav\", format=\"wav\")\n        except:\n            print(path)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:19.168159Z","iopub.execute_input":"2023-07-18T10:50:19.168590Z","iopub.status.idle":"2023-07-18T10:50:19.177168Z","shell.execute_reply.started":"2023-07-18T10:50:19.168550Z","shell.execute_reply":"2023-07-18T10:50:19.175707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\nnum_cores = multiprocessing.cpu_count()\nprint(\"Number of CPUs:\", num_cores)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:19.179114Z","iopub.execute_input":"2023-07-18T10:50:19.179642Z","iopub.status.idle":"2023-07-18T10:50:19.206902Z","shell.execute_reply.started":"2023-07-18T10:50:19.179578Z","shell.execute_reply":"2023-07-18T10:50:19.205879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nstart = time.time()\n\nParallel(n_jobs=num_cores, backend=\"multiprocessing\")(\n    delayed(save_fn)(filename) for filename in tqdm(audio_files)\n)\n\nend = time.time()\nprint(\"total time to process: {x} seconds\".format(x=end-start))","metadata":{"execution":{"iopub.status.busy":"2023-07-18T10:50:19.208075Z","iopub.execute_input":"2023-07-18T10:50:19.208453Z","iopub.status.idle":"2023-07-18T10:51:39.265381Z","shell.execute_reply.started":"2023-07-18T10:50:19.208389Z","shell.execute_reply":"2023-07-18T10:51:39.263506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}