{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install transformers\n!pip install jiwer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-15T14:21:47.373978Z","iopub.execute_input":"2023-09-15T14:21:47.374697Z","iopub.status.idle":"2023-09-15T14:22:12.785437Z","shell.execute_reply.started":"2023-09-15T14:21:47.374651Z","shell.execute_reply":"2023-09-15T14:22:12.784206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall -y torchaudio","metadata":{"execution":{"iopub.status.busy":"2023-09-15T14:29:02.455348Z","iopub.execute_input":"2023-09-15T14:29:02.455852Z","iopub.status.idle":"2023-09-15T14:29:04.757093Z","shell.execute_reply.started":"2023-09-15T14:29:02.455808Z","shell.execute_reply":"2023-09-15T14:29:04.75594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!add-apt-repository -y ppa:savoury1/ffmpeg4\n!apt-get -qq install -y ffmpeg","metadata":{"execution":{"iopub.status.busy":"2023-09-15T14:29:04.759355Z","iopub.execute_input":"2023-09-15T14:29:04.76004Z","iopub.status.idle":"2023-09-15T14:29:13.568949Z","shell.execute_reply.started":"2023-09-15T14:29:04.759995Z","shell.execute_reply":"2023-09-15T14:29:13.56763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U datasets>=2.14.5","metadata":{"execution":{"iopub.status.busy":"2023-09-15T14:30:27.298991Z","iopub.execute_input":"2023-09-15T14:30:27.299411Z","iopub.status.idle":"2023-09-15T14:30:31.986065Z","shell.execute_reply.started":"2023-09-15T14:30:27.299376Z","shell.execute_reply":"2023-09-15T14:30:31.984072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom datasets import Dataset, DatasetDict, Audio","metadata":{"execution":{"iopub.status.busy":"2023-09-15T16:02:56.139649Z","iopub.execute_input":"2023-09-15T16:02:56.140028Z","iopub.status.idle":"2023-09-15T16:02:56.926238Z","shell.execute_reply.started":"2023-09-15T16:02:56.139998Z","shell.execute_reply":"2023-09-15T16:02:56.925293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_custom_dataset(audio_folder, sentences_csv):\n    sentences_df = pd.read_csv(sentences_csv)\n    sentences_df = sentences_df[10:]\n    print(sentences_df.head())\n    train_paths = sentences_df['id'].apply(lambda x: str(os.path.join(audio_folder,  x + \".mp3\")))\n    print(train_paths)\n    train_dataset = Dataset.from_dict({\"audio\":train_paths.tolist() ,\"sentence\": sentences_df['sentence'].tolist()}).cast_column(\"audio\", Audio(sampling_rate=16_000))\n    return train_dataset\n \n# Specify the paths \naudio_folder = '/kaggle/input/bengaliai-speech/train_mp3s'\nsentences_csv = '/kaggle/input/bengaliai-speech/train.csv'\n\n# Create the custom DatasetDict\ntrain_dataset = create_custom_dataset(audio_folder, sentences_csv)","metadata":{"execution":{"iopub.status.busy":"2023-09-15T16:07:13.591073Z","iopub.execute_input":"2023-09-15T16:07:13.591497Z","iopub.status.idle":"2023-09-15T16:07:19.212673Z","shell.execute_reply.started":"2023-09-15T16:07:13.591464Z","shell.execute_reply":"2023-09-15T16:07:19.208639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.","metadata":{"execution":{"iopub.status.busy":"2023-09-15T16:04:56.359517Z","iopub.execute_input":"2023-09-15T16:04:56.360624Z","iopub.status.idle":"2023-09-15T16:04:56.370233Z","shell.execute_reply.started":"2023-09-15T16:04:56.360584Z","shell.execute_reply":"2023-09-15T16:04:56.369141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.to_csv(\"/kaggle/working/train_10_mp3s.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-09-15T16:04:58.707075Z","iopub.execute_input":"2023-09-15T16:04:58.707482Z","iopub.status.idle":"2023-09-15T16:05:11.006077Z","shell.execute_reply.started":"2023-09-15T16:04:58.70745Z","shell.execute_reply":"2023-09-15T16:05:11.004981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/train_10_mp3s.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-15T16:05:15.331297Z","iopub.execute_input":"2023-09-15T16:05:15.331713Z","iopub.status.idle":"2023-09-15T16:05:19.141081Z","shell.execute_reply.started":"2023-09-15T16:05:15.331682Z","shell.execute_reply":"2023-09-15T16:05:19.140006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport skimage.io\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport pandas as pd\nimport numpy as np\nimport shutil\n\nfrom pydub import AudioSegment\nfrom joblib import Parallel, delayed\n\ndf = pd.read_csv(\"../input/bengaliai-speech/train.csv\")\ndf.shape\n\nROOT_PATH = \"../input/bengaliai-speech/train_mp3s\"\nOUTPUT_DIR = \"../input/train_files_wav\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\ndf['folder'] = df['id'].apply(lambda x: x[0])\ndf['folder1'] = df['id'].apply(lambda x: x[1])\n\nfolders = sorted(list(set(df['folder'].tolist())))\nfolders1 = sorted(list(set(df['folder1'].tolist())))\nprint(\"Folders\", folders, folders1)\nprint(\"Total Folders\", len(folders), len(folders1))\nfor folder in folders:\n    if folder in ['0','1','2','3','4','5']:\n        continue\n    for folder1 in folders:\n        print(folder, folder1, \"Started Folder\")\n        audio_files = df[((df['folder'] == folder) & (df['folder1'] == folder1))]['id'].apply(lambda x:x+\".mp3\").tolist()\n        def save_fn(filename):\n            folder = filename[0:1]\n            path = f\"{ROOT_PATH}/{filename}\"\n            save_path = f\"{OUTPUT_DIR}/{folder}/wav/{folder1}\"\n            if not os.path.exists(save_path):\n                os.makedirs(save_path, exist_ok=True)\n\n            if os.path.exists(path):\n                try:\n                    sound = AudioSegment.from_mp3(path)\n                    sound.export(f\"{save_path}/{filename[:-4]}.wav\", format=\"wav\")\n                except:\n                    print(path)\n\n        import multiprocessing\n        num_cores = multiprocessing.cpu_count()\n\n        import time\n        start = time.time()\n\n        Parallel(n_jobs=72, backend=\"multiprocessing\")(\n            delayed(save_fn)(filename) for filename in tqdm(audio_files)\n        )\n\n        end = time.time()\n        print(folder, folder1, \"total time to process: {x} seconds\".format(x=end-start))","metadata":{"execution":{"iopub.status.busy":"2023-09-13T20:53:31.102035Z","iopub.execute_input":"2023-09-13T20:53:31.102395Z","iopub.status.idle":"2023-09-13T20:53:31.293044Z","shell.execute_reply.started":"2023-09-13T20:53:31.102365Z","shell.execute_reply":"2023-09-13T20:53:31.292062Z"},"trusted":true},"execution_count":null,"outputs":[]}]}