{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In my previous [notebook](https://www.kaggle.com/code/mbmmurad/mp3-to-wav-conversion), we tried to convert mp3 audio files to wav files. But it took almost 25 minutes to convert the Validation files. In this notebook we'll try a different way to convert the mp3 files to wav files using joblib. This notebook takes only 8-9 minutes to process the validation files. \n(3times faster)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T16:19:04.757196Z","iopub.execute_input":"2022-07-02T16:19:04.757537Z","iopub.status.idle":"2022-07-02T16:19:04.764002Z","shell.execute_reply.started":"2022-07-02T16:19:04.757508Z","shell.execute_reply":"2022-07-02T16:19:04.762854Z"}}},{"cell_type":"markdown","source":"**Reference** :\n\nIdea is from this notebook : https://www.kaggle.com/code/kingabzpro/asr-mp3-to-wav-dataset?fbclid=IwAR3D61p3PW9lY-IGTOLrjf52b1_SQkc2FFuHl9kEbGumVlgHLqVOGJoQl5w","metadata":{}},{"cell_type":"markdown","source":"# Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport skimage.io\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport pandas as pd\nimport numpy as np\nimport shutil\n\nfrom pydub import AudioSegment\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:35:33.396183Z","iopub.execute_input":"2022-07-06T15:35:33.396559Z","iopub.status.idle":"2022-07-06T15:35:34.521605Z","shell.execute_reply.started":"2022-07-06T15:35:33.396475Z","shell.execute_reply":"2022-07-06T15:35:34.520485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_PATH = \"../input/dlsprint/train_files\"\nOUTPUT_DIR = \"./train_files_wav\"","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:35:47.594095Z","iopub.execute_input":"2022-07-06T15:35:47.594521Z","iopub.status.idle":"2022-07-06T15:35:47.599049Z","shell.execute_reply.started":"2022-07-06T15:35:47.59449Z","shell.execute_reply":"2022-07-06T15:35:47.598023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(OUTPUT_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:35:51.087166Z","iopub.execute_input":"2022-07-06T15:35:51.087567Z","iopub.status.idle":"2022-07-06T15:35:51.092585Z","shell.execute_reply.started":"2022-07-06T15:35:51.087534Z","shell.execute_reply":"2022-07-06T15:35:51.09172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_fn(filename):\n    \n    path = f\"{ROOT_PATH}/{filename}\"\n    save_path = f\"{OUTPUT_DIR}\"\n    if not os.path.exists(save_path):\n        os.makedirs(save_path, exist_ok=True)\n    \n    if os.path.exists(path):\n        try:\n            sound = AudioSegment.from_mp3(path)\n            sound = sound.set_frame_rate(16000)\n            sound.export(f\"{save_path}/{filename[:-4]}.wav\", format=\"wav\")\n        except:\n            print(path)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:35:51.670528Z","iopub.execute_input":"2022-07-06T15:35:51.670947Z","iopub.status.idle":"2022-07-06T15:35:51.677025Z","shell.execute_reply.started":"2022-07-06T15:35:51.67091Z","shell.execute_reply":"2022-07-06T15:35:51.676023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#-------------------------------\n# imports\n#-------------------------------\nimport os \nimport pandas as pd \nfrom tqdm.auto import tqdm\nimport warnings\nimport librosa\nimport io\nimport soundfile as sf\ntqdm.pandas()\nwarnings.filterwarnings('ignore')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:36:33.084283Z","iopub.execute_input":"2022-07-06T15:36:33.085484Z","iopub.status.idle":"2022-07-06T15:36:34.748981Z","shell.execute_reply.started":"2022-07-06T15:36:33.08544Z","shell.execute_reply":"2022-07-06T15:36:34.747571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#---------------\n# data filtering\n#---------------\ndef filter_votes(x):\n    up=x[\"up_votes\"]\n    down=x[\"down_votes\"]\n    if up-down<=0:\n        return None\n    elif up==0:\n        return None\n    else:\n        return up\n# ------------------------- train data----------------------------------------\ntrain_df=pd.read_csv(\"../input/dlsprint/train.csv\")\nprint(\"Total Data before filtering:\",len(train_df))\ntrain_df[\"up_votes\"]=train_df.progress_apply(lambda x:filter_votes(x),axis=1)\ntrain_df.dropna(subset = ['up_votes'],inplace=True)\nprint(\"Total Data after filtering:\",len(train_df))\naudio_files=train_df[\"path\"].tolist()\nlen(audio_files)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:38:14.888408Z","iopub.execute_input":"2022-07-06T15:38:14.888793Z","iopub.status.idle":"2022-07-06T15:38:20.148079Z","shell.execute_reply.started":"2022-07-06T15:38:14.888762Z","shell.execute_reply":"2022-07-06T15:38:20.146808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nstart = time.time()\n\nParallel(n_jobs=8, backend=\"multiprocessing\")(\n    delayed(save_fn)(filename) for filename in tqdm(audio_files)\n)\n\nend = time.time()\nprint(\"total time to process: {x} seconds\".format(x=end-start))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:38:42.992828Z","iopub.execute_input":"2022-07-06T15:38:42.993227Z","iopub.status.idle":"2022-07-06T16:20:22.419263Z","shell.execute_reply.started":"2022-07-06T15:38:42.993185Z","shell.execute_reply":"2022-07-06T16:20:22.416876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}