{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom datasets import Dataset, Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-28T20:29:53.214256Z","iopub.execute_input":"2023-08-28T20:29:53.214703Z","iopub.status.idle":"2023-08-28T20:29:54.175349Z","shell.execute_reply.started":"2023-08-28T20:29:53.214666Z","shell.execute_reply":"2023-08-28T20:29:54.174339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CV13_DIR = \"/kaggle/input/common-voice-13-bengali-normalized\"","metadata":{"execution":{"iopub.status.busy":"2023-08-28T20:29:48.580291Z","iopub.execute_input":"2023-08-28T20:29:48.580669Z","iopub.status.idle":"2023-08-28T20:29:48.585488Z","shell.execute_reply.started":"2023-08-28T20:29:48.580639Z","shell.execute_reply":"2023-08-28T20:29:48.584409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(os.path.join(CV13_DIR, \"train.tsv\"), sep=\"\\t\")\ndf[\"path\"] = df[\"path\"].apply(lambda path: os.path.join(CV13_DIR, \"train\", path))\ndf = df.rename(columns={\"path\": \"audio\"})\n\ndataset = Dataset.from_pandas(df).cast_column(\"audio\", Audio(sampling_rate=16_000))","metadata":{"execution":{"iopub.status.busy":"2023-08-28T20:34:47.965759Z","iopub.execute_input":"2023-08-28T20:34:47.966183Z","iopub.status.idle":"2023-08-28T20:34:48.183735Z","shell.execute_reply.started":"2023-08-28T20:34:47.966146Z","shell.execute_reply":"2023-08-28T20:34:48.182891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoProcessor\n\nprocessor = AutoProcessor.from_pretrained(\"Umong/wav2vec2-large-mms-1b-bengali\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T20:36:34.776193Z","iopub.execute_input":"2023-08-28T20:36:34.776924Z","iopub.status.idle":"2023-08-28T20:36:37.345011Z","shell.execute_reply.started":"2023-08-28T20:36:34.776888Z","shell.execute_reply":"2023-08-28T20:36:37.343762Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(sample):\n    audio = sample[\"audio\"]\n    sample[\"input_values\"] = processor(audio[\"array\"], sampling_rate=audio[\"sampling_rate\"]).input_values[0]\n    sample[\"input_length\"] = len(sample[\"input_values\"])\n\n    sample[\"labels\"] = processor(text=sample[\"sentence\"]).input_ids\n    return sample","metadata":{"execution":{"iopub.status.busy":"2023-08-28T21:06:09.524492Z","iopub.execute_input":"2023-08-28T21:06:09.524941Z","iopub.status.idle":"2023-08-28T21:06:09.532116Z","shell.execute_reply.started":"2023-08-28T21:06:09.524904Z","shell.execute_reply":"2023-08-28T21:06:09.530702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.map(prepare_dataset, remove_columns=dataset.column_names)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T20:50:52.675098Z","iopub.execute_input":"2023-08-28T20:50:52.675542Z","iopub.status.idle":"2023-08-28T20:55:42.229824Z","shell.execute_reply.started":"2023-08-28T20:50:52.675504Z","shell.execute_reply":"2023-08-28T20:55:42.228530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset[0][\"labels\"]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T21:00:10.336287Z","iopub.execute_input":"2023-08-28T21:00:10.336774Z","iopub.status.idle":"2023-08-28T21:00:10.353609Z","shell.execute_reply.started":"2023-08-28T21:00:10.336728Z","shell.execute_reply":"2023-08-28T21:00:10.352420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# or apply the transformation on-the-fly \n\ndef prepare_dataset(batch):\n    batch[\"input_values\"] = []\n    batch[\"input_length\"] = []\n    \n    for audio in batch[\"audio\"]:\n        input_values = processor(audio[\"array\"], sampling_rate=audio[\"sampling_rate\"]).input_values[0]\n        input_length = len(input_values)\n        \n        batch[\"input_values\"].append(input_values)\n        batch[\"input_length\"].append(input_length)\n    \n\n    batch[\"labels\"] = processor(text=batch[\"sentence\"]).input_ids\n    return batch\n\ndataset = Dataset.from_pandas(df).cast_column(\"audio\", Audio(sampling_rate=16_000))\ndataset.set_transform(prepare_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T21:05:38.332903Z","iopub.execute_input":"2023-08-28T21:05:38.334649Z","iopub.status.idle":"2023-08-28T21:05:38.379699Z","shell.execute_reply.started":"2023-08-28T21:05:38.334591Z","shell.execute_reply":"2023-08-28T21:05:38.377671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset[0][\"labels\"]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T21:05:50.614592Z","iopub.execute_input":"2023-08-28T21:05:50.614990Z","iopub.status.idle":"2023-08-28T21:05:50.630379Z","shell.execute_reply.started":"2023-08-28T21:05:50.614958Z","shell.execute_reply":"2023-08-28T21:05:50.629044Z"},"trusted":true},"execution_count":null,"outputs":[]}]}