{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from os import path\nimport os\nfrom pydub import AudioSegment","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-28T08:49:07.792802Z","iopub.execute_input":"2023-08-28T08:49:07.793198Z","iopub.status.idle":"2023-08-28T08:49:07.799289Z","shell.execute_reply.started":"2023-08-28T08:49:07.793170Z","shell.execute_reply":"2023-08-28T08:49:07.797774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if os.path.exists('train') == False:\n    os.makedirs('train')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:08.888296Z","iopub.execute_input":"2023-08-28T08:49:08.888792Z","iopub.status.idle":"2023-08-28T08:49:08.894699Z","shell.execute_reply.started":"2023-08-28T08:49:08.888749Z","shell.execute_reply":"2023-08-28T08:49:08.893682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain_df = pd.read_csv('/kaggle/input/speech-metadata/shuffle_train.csv')\ntrain_ids = train_df['id'].tolist()\ntrain_ids = train_ids[:70000]\ndata_path: str = \"/kaggle/input/bengaliai-speech/train_mp3s\"       \ntrain_paths = [f'{data_path}/{train_id}.mp3' for train_id in train_ids]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T09:01:18.192962Z","iopub.execute_input":"2023-08-28T09:01:18.193456Z","iopub.status.idle":"2023-08-28T09:01:22.333014Z","shell.execute_reply.started":"2023-08-28T09:01:18.193417Z","shell.execute_reply":"2023-08-28T09:01:22.331310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nimport multiprocessing as mp\nfrom multiprocessing import cpu_count\ncpu_count()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T09:01:24.910161Z","iopub.execute_input":"2023-08-28T09:01:24.910614Z","iopub.status.idle":"2023-08-28T09:01:24.919653Z","shell.execute_reply.started":"2023-08-28T09:01:24.910578Z","shell.execute_reply":"2023-08-28T09:01:24.918299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(train_id):\n    src_path = f'{data_path}/{train_id}.mp3'\n    dst_path = f'train/{train_id}.wav'\n    sound = AudioSegment.from_mp3(src_path)\n    sound.export(dst_path, format=\"wav\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T09:01:41.707801Z","iopub.execute_input":"2023-08-28T09:01:41.708353Z","iopub.status.idle":"2023-08-28T09:01:41.715589Z","shell.execute_reply.started":"2023-08-28T09:01:41.708309Z","shell.execute_reply":"2023-08-28T09:01:41.714333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n_ = Parallel(n_jobs=cpu_count())(\n    delayed(process)(train_id)\n    for train_id in tqdm(train_ids)\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T09:02:29.273165Z","iopub.execute_input":"2023-08-28T09:02:29.273657Z","iopub.status.idle":"2023-08-28T09:02:57.617003Z","shell.execute_reply.started":"2023-08-28T09:02:29.273610Z","shell.execute_reply":"2023-08-28T09:02:57.614839Z"},"trusted":true},"execution_count":null,"outputs":[]}]}