{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install transformers\n!pip install jiwer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-31T15:18:08.875903Z","iopub.execute_input":"2023-08-31T15:18:08.876251Z","iopub.status.idle":"2023-08-31T15:18:32.136109Z","shell.execute_reply.started":"2023-08-31T15:18:08.876223Z","shell.execute_reply":"2023-08-31T15:18:32.134890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom datasets import Dataset, DatasetDict, Audio\nfrom transformers import WhisperFeatureExtractor, WhisperProcessor, WhisperTokenizer","metadata":{"execution":{"iopub.status.busy":"2023-08-31T15:18:32.142504Z","iopub.execute_input":"2023-08-31T15:18:32.142837Z","iopub.status.idle":"2023-08-31T15:18:33.401897Z","shell.execute_reply.started":"2023-08-31T15:18:32.142796Z","shell.execute_reply":"2023-08-31T15:18:33.400756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_custom_dataset(audio_folder, sentences_csv):\n    \n    sentences_df = pd.read_csv(sentences_csv)\n    audio_paths = sorted([os.path.join(audio_folder, filename) for filename in os.listdir(audio_folder)])\n    sentence_list = sentences_df['sentence'].tolist()\n    \n    audio_dataset = Dataset.from_dict({\"audio\": audio_paths, \"sentence\": sentence_list}).cast_column(\"audio\", Audio(sampling_rate=16000))\n    train_dataset = DatasetDict({\"train\": audio_dataset})\n\n    return train_dataset\n\n# Specify the paths \naudio_folder = '/kaggle/input/bengaliai-speech/train_mp3s'\nsentences_csv = '/kaggle/input/bengaliai-speech/train.csv'\n\n# Create the custom DatasetDict\ntrain_dataset = create_custom_dataset(audio_folder, sentences_csv)","metadata":{"execution":{"iopub.status.busy":"2023-08-31T15:18:33.403285Z","iopub.execute_input":"2023-08-31T15:18:33.403936Z","iopub.status.idle":"2023-08-31T15:18:42.814699Z","shell.execute_reply.started":"2023-08-31T15:18:33.403899Z","shell.execute_reply":"2023-08-31T15:18:42.813546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset","metadata":{"execution":{"iopub.status.busy":"2023-08-31T15:18:42.817336Z","iopub.execute_input":"2023-08-31T15:18:42.818142Z","iopub.status.idle":"2023-08-31T15:18:42.826664Z","shell.execute_reply.started":"2023-08-31T15:18:42.818107Z","shell.execute_reply":"2023-08-31T15:18:42.825546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Everytime I want to check the audio, an error occurs. would love for a bit of help in the comments","metadata":{}},{"cell_type":"code","source":"train_dataset['train'][0]","metadata":{"execution":{"iopub.status.busy":"2023-08-31T15:18:42.828056Z","iopub.execute_input":"2023-08-31T15:18:42.828445Z","iopub.status.idle":"2023-08-31T15:18:45.843690Z","shell.execute_reply.started":"2023-08-31T15:18:42.828396Z","shell.execute_reply":"2023-08-31T15:18:45.841960Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}