{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":9022117,"sourceType":"datasetVersion","datasetId":5436906}],"dockerImageVersionId":30528,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!pip install evaluate\n!pip install jiwer","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:11:10.919009Z","iopub.execute_input":"2024-07-24T19:11:10.919291Z","iopub.status.idle":"2024-07-24T19:11:34.975369Z","shell.execute_reply.started":"2024-07-24T19:11:10.919264Z","shell.execute_reply":"2024-07-24T19:11:34.974281Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import the necessary libraries","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom IPython.display import Audio\nimport librosa  \nfrom transformers import WhisperProcessor\nimport torch\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nimport evaluate\nfrom transformers.models.whisper.english_normalizer import BasicTextNormalizer\nfrom transformers import WhisperForConditionalGeneration\nfrom functools import partial\nfrom transformers import Seq2SeqTrainingArguments\nfrom transformers import Seq2SeqTrainer\nfrom transformers.models.whisper.tokenization_whisper import TO_LANGUAGE_CODE\n\nTO_LANGUAGE_CODE['bengali']","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:11:34.977198Z","iopub.execute_input":"2024-07-24T19:11:34.977509Z","iopub.status.idle":"2024-07-24T19:11:49.526460Z","shell.execute_reply.started":"2024-07-24T19:11:34.977479Z","shell.execute_reply":"2024-07-24T19:11:49.525519Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the dataset","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/bengaliai-speech/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:11:49.527969Z","iopub.execute_input":"2024-07-24T19:11:49.528308Z","iopub.status.idle":"2024-07-24T19:11:49.537412Z","shell.execute_reply.started":"2024-07-24T19:11:49.528274Z","shell.execute_reply":"2024-07-24T19:11:49.536553Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/train-csv-new/new_data_for_train.csv')\ntrain['path']= path + '/train_mp3s/' + train['id']+'.mp3'\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:11:49.539354Z","iopub.execute_input":"2024-07-24T19:11:49.539671Z","iopub.status.idle":"2024-07-24T19:11:57.471436Z","shell.execute_reply.started":"2024-07-24T19:11:49.539642Z","shell.execute_reply":"2024-07-24T19:11:57.470528Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_rate = 22500  # this is a common sample rate for audio\n\n# Load the audio file using librosa or any other audio processing library you prefer\naudio_path=path + '/train_mp3s/' + train['id'].iloc[7]+'.mp3'\naudio_data, _ = librosa.load(audio_path, sr=sample_rate)\nprint(train['sentence'].iloc[7])\n# Display the audio using IPython.display.Audio\ndisplay(Audio(data=audio_data, rate=sample_rate))","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:12:03.344000Z","iopub.execute_input":"2024-07-24T19:12:03.344349Z","iopub.status.idle":"2024-07-24T19:12:12.381639Z","shell.execute_reply.started":"2024-07-24T19:12:03.344319Z","shell.execute_reply":"2024-07-24T19:12:12.380789Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the model and preprocessor","metadata":{}},{"cell_type":"code","source":"model_id = \"openai/whisper-small\"\n\nprocessor = WhisperProcessor.from_pretrained(\n    model_id, \n    language=\"bengali\", \n    task=\"transcribe\"\n)\n\nprocessor","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:12:20.465839Z","iopub.execute_input":"2024-07-24T19:12:20.467117Z","iopub.status.idle":"2024-07-24T19:12:23.165447Z","shell.execute_reply.started":"2024-07-24T19:12:20.467079Z","shell.execute_reply":"2024-07-24T19:12:23.164559Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampling_rate = processor.feature_extractor.sampling_rate\nsampling_rate","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:12:27.357902Z","iopub.execute_input":"2024-07-24T19:12:27.358729Z","iopub.status.idle":"2024-07-24T19:12:27.364336Z","shell.execute_reply.started":"2024-07-24T19:12:27.358694Z","shell.execute_reply":"2024-07-24T19:12:27.363456Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_path","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:11.798504Z","iopub.execute_input":"2024-07-24T19:13:11.799606Z","iopub.status.idle":"2024-07-24T19:13:11.805257Z","shell.execute_reply.started":"2024-07-24T19:13:11.799571Z","shell.execute_reply":"2024-07-24T19:13:11.804326Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_audio(audio_path):\n    audio_arrays, sampling_rate = librosa.load(audio_path)\n    return audio_arrays, sampling_rate\n\nload_audio(audio_path)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:12:34.248835Z","iopub.execute_input":"2024-07-24T19:12:34.249542Z","iopub.status.idle":"2024-07-24T19:12:34.269003Z","shell.execute_reply.started":"2024-07-24T19:12:34.249512Z","shell.execute_reply":"2024-07-24T19:12:34.268087Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- {'audio': Audio(sampling_rate=48000, mono=True, decode=True, id=None),\n-  'sentence': Value(dtype='string', id=None)}","metadata":{}},{"cell_type":"code","source":"processor.feature_extractor.model_input_names\naudio_arrays, _= librosa.load(train['path'].iloc[1], sr=sample_rate)\naudio_arrays.tolist()[0]","metadata":{"execution":{"iopub.status.busy":"2024-07-24T12:26:31.463019Z","iopub.execute_input":"2024-07-24T12:26:31.463364Z","iopub.status.idle":"2024-07-24T12:26:31.490610Z","shell.execute_reply.started":"2024-07-24T12:26:31.463336Z","shell.execute_reply":"2024-07-24T12:26:31.489680Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_dataset(example, sample_rate= sample_rate):\n    audio_arrays, sampling_rate = librosa.load(example['path'], sr=sample_rate)\n    \n    example = processor(\n        audio=audio_arrays,\n        sampling_rate=sampling_rate,\n        text=example[\"ipas\"],\n    )\n    # compute input length of audio sample in seconds\n    example[\"input_length\"] = len(audio_arrays) / sampling_rate\n\n    return example","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:25.992866Z","iopub.execute_input":"2024-07-24T19:13:25.993464Z","iopub.status.idle":"2024-07-24T19:13:25.998899Z","shell.execute_reply.started":"2024-07-24T19:13:25.993435Z","shell.execute_reply":"2024-07-24T19:13:25.997892Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_rate = processor.feature_extractor.sampling_rate\n\ninputs = prepare_dataset(train.iloc[0], sample_rate= sample_rate)\nprint('Audio files  :',inputs['input_features'])\nprint('Text vector  :',inputs['labels'])\nprint('Input length :',inputs['input_length'])","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:27.535949Z","iopub.execute_input":"2024-07-24T19:13:27.536670Z","iopub.status.idle":"2024-07-24T19:13:27.615729Z","shell.execute_reply.started":"2024-07-24T19:13:27.536639Z","shell.execute_reply":"2024-07-24T19:13:27.614530Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decoded_text = processor.decode(inputs['labels'])\nprint(decoded_text)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:31.683581Z","iopub.execute_input":"2024-07-24T19:13:31.684418Z","iopub.status.idle":"2024-07-24T19:13:31.689469Z","shell.execute_reply.started":"2024-07-24T19:13:31.684386Z","shell.execute_reply":"2024-07-24T19:13:31.688574Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Audio\n\n# Convert the encoded audio back to waveform using librosa\ndecoded_audio = processor.feature_extractor.sampling_rate * librosa.feature.inverse.mel_to_audio(\n    inputs['input_features'][0], sr=processor.feature_extractor.sampling_rate\n)\n\n# Display the audio using IPython.display.Audio\ndisplay(Audio(decoded_audio, rate=sample_rate))","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:35.649425Z","iopub.execute_input":"2024-07-24T19:13:35.650239Z","iopub.status.idle":"2024-07-24T19:13:54.888148Z","shell.execute_reply.started":"2024-07-24T19:13:35.650206Z","shell.execute_reply":"2024-07-24T19:13:54.886543Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \ntrain_data = []\nfor _, row in tqdm(train.iloc[:20000].iterrows()):\n    train_data.append(prepare_dataset(row, sample_rate=sample_rate))","metadata":{"execution":{"iopub.status.busy":"2024-07-24T12:28:12.588062Z","iopub.execute_input":"2024-07-24T12:28:12.588796Z","iopub.status.idle":"2024-07-24T13:09:58.751607Z","shell.execute_reply.started":"2024-07-24T12:28:12.588761Z","shell.execute_reply":"2024-07-24T13:09:58.750045Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \nval_data = []\nfor _, row in tqdm(train.iloc[20000:22000].iterrows()):\n    val_data.append(prepare_dataset(row, sample_rate=sample_rate))","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:13:54.890662Z","iopub.execute_input":"2024-07-24T19:13:54.890985Z","iopub.status.idle":"2024-07-24T19:18:14.476727Z","shell.execute_reply.started":"2024-07-24T19:13:54.890955Z","shell.execute_reply":"2024-07-24T19:18:14.475527Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_data","metadata":{"execution":{"iopub.status.busy":"2024-07-24T19:19:25.342428Z","iopub.execute_input":"2024-07-24T19:19:25.342828Z","iopub.status.idle":"2024-07-24T19:19:25.950337Z","shell.execute_reply.started":"2024-07-24T19:19:25.342794Z","shell.execute_reply":"2024-07-24T19:19:25.949466Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(\n        self, features: List[Dict[str, Union[List[int], torch.Tensor]]]\n    ) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [\n            {\"input_features\": feature[\"input_features\"][0]} for feature in features\n        ]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(\n            labels_batch.attention_mask.ne(1), -100\n        )\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n\n        return batch\n    \ndata_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:33.580713Z","iopub.execute_input":"2024-07-24T13:18:33.581442Z","iopub.status.idle":"2024-07-24T13:18:33.590796Z","shell.execute_reply.started":"2024-07-24T13:18:33.581408Z","shell.execute_reply":"2024-07-24T13:18:33.589805Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metric = evaluate.load(\"wer\")","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:38.202426Z","iopub.execute_input":"2024-07-24T13:18:38.203479Z","iopub.status.idle":"2024-07-24T13:18:38.945257Z","shell.execute_reply.started":"2024-07-24T13:18:38.203441Z","shell.execute_reply":"2024-07-24T13:18:38.944338Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = BasicTextNormalizer()\n\n\ndef compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    # replace -100 with the pad_token_id\n    label_ids[label_ids == -100] = processor.tokenizer.pad_token_id\n\n    # we do not want to group tokens when computing the metrics\n    pred_str = processor.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = processor.batch_decode(label_ids, skip_special_tokens=True)\n\n    # compute orthographic wer\n    wer_ortho = 100 * metric.compute(predictions=pred_str, references=label_str)\n\n    # compute normalised WER\n    pred_str_norm = [normalizer(pred) for pred in pred_str]\n    label_str_norm = [normalizer(label) for label in label_str]\n    # filtering step to only evaluate the samples that correspond to non-zero references:\n    pred_str_norm = [\n        pred_str_norm[i] for i in range(len(pred_str_norm)) if len(label_str_norm[i]) > 0\n    ]\n    label_str_norm = [\n        label_str_norm[i]\n        for i in range(len(label_str_norm))\n        if len(label_str_norm[i]) > 0\n    ]\n\n    wer = 100 * metric.compute(predictions=pred_str_norm, references=label_str_norm)\n\n    return {\"wer_ortho\": wer_ortho, \"wer\": wer}","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:46.220793Z","iopub.execute_input":"2024-07-24T13:18:46.221655Z","iopub.status.idle":"2024-07-24T13:18:46.230170Z","shell.execute_reply.started":"2024-07-24T13:18:46.221606Z","shell.execute_reply":"2024-07-24T13:18:46.229283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = WhisperForConditionalGeneration.from_pretrained(\"openai/whisper-small\")\n\n# disable cache during training since it's incompatible with gradient checkpointing\nmodel.config.use_cache = False\n\n# set language and task for generation and re-enable cache\nmodel.generate = partial(\n    model.generate, language=\"bengali\", task=\"transcribe\", use_cache=True\n)\n\nmodel","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:49.348000Z","iopub.execute_input":"2024-07-24T13:18:49.348364Z","iopub.status.idle":"2024-07-24T13:18:56.207262Z","shell.execute_reply.started":"2024-07-24T13:18:49.348334Z","shell.execute_reply":"2024-07-24T13:18:56.206408Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_args = Seq2SeqTrainingArguments(\n    output_dir=\"whisper-small-dv\", \n    per_device_train_batch_size=16,\n    gradient_accumulation_steps=1,  # increase by 2x for every 2x decrease in batch size\n    learning_rate=1e-5,\n    lr_scheduler_type=\"constant_with_warmup\",\n    warmup_steps=5,\n    max_steps=500,\n    gradient_checkpointing=True,\n    fp16=True,\n    fp16_full_eval=True,\n    evaluation_strategy=\"steps\",\n    per_device_eval_batch_size=16,\n    predict_with_generate=True,\n    generation_max_length=225,\n    save_steps=500,\n    eval_steps=500,\n    logging_steps=25,\n    report_to=[\"tensorboard\"],\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    #push_to_hub=True,\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:56.209059Z","iopub.execute_input":"2024-07-24T13:18:56.209504Z","iopub.status.idle":"2024-07-24T13:18:56.365116Z","shell.execute_reply.started":"2024-07-24T13:18:56.209469Z","shell.execute_reply":"2024-07-24T13:18:56.364364Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=train_data,\n    eval_dataset=val_data,\n    data_collator=data_collator,\n    compute_metrics=compute_metrics,\n    tokenizer=processor,\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:18:56.366145Z","iopub.execute_input":"2024-07-24T13:18:56.366424Z","iopub.status.idle":"2024-07-24T13:18:59.176690Z","shell.execute_reply.started":"2024-07-24T13:18:56.366397Z","shell.execute_reply":"2024-07-24T13:18:59.175686Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2024-07-24T13:19:05.561014Z","iopub.execute_input":"2024-07-24T13:19:05.561352Z","iopub.status.idle":"2024-07-24T14:21:04.345390Z","shell.execute_reply.started":"2024-07-24T13:19:05.561325Z","shell.execute_reply":"2024-07-24T14:21:04.344435Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.save_model('whisper-small-dv')","metadata":{"execution":{"iopub.status.busy":"2024-07-24T14:21:04.347438Z","iopub.execute_input":"2024-07-24T14:21:04.348093Z","iopub.status.idle":"2024-07-24T14:21:07.719475Z","shell.execute_reply.started":"2024-07-24T14:21:04.348057Z","shell.execute_reply":"2024-07-24T14:21:07.718662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport subprocess\nfrom IPython.display import FileLink, display\n\ndef download_file(path, download_file_name):\n    os.chdir('/kaggle/working/')\n    zip_name = f\"/kaggle/working/{download_file_name}.zip\"\n    command = f\"zip {zip_name} {path} -r\"\n    result = subprocess.run(command, shell=True, capture_output=True, text=True)\n    if result.returncode != 0:\n        print(\"Unable to run zip command!\")\n        print(result.stderr)\n        return\n    display(FileLink(f'{download_file_name}.zip'))","metadata":{"execution":{"iopub.status.busy":"2024-07-24T14:21:07.720517Z","iopub.execute_input":"2024-07-24T14:21:07.720818Z","iopub.status.idle":"2024-07-24T14:21:07.726998Z","shell.execute_reply.started":"2024-07-24T14:21:07.720792Z","shell.execute_reply":"2024-07-24T14:21:07.726097Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"download_file(\"/kaggle/working/whisper-small-dv\", f\"whisper-small-dv_1_submission\")","metadata":{"execution":{"iopub.status.busy":"2024-07-24T14:23:00.980315Z","iopub.execute_input":"2024-07-24T14:23:00.980708Z","iopub.status.idle":"2024-07-24T14:26:20.337689Z","shell.execute_reply.started":"2024-07-24T14:23:00.980677Z","shell.execute_reply":"2024-07-24T14:26:20.336694Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the directory where you want to save the model\n#save_directory = \"/kaggle/working/\"\n\n# Save the model and tokenizer\nmodel.save_pretrained()\nprocessor.save_pretrained()\n\nprint(\"Model and tokenizer saved to:\", save_directory)","metadata":{"execution":{"iopub.status.busy":"2024-07-24T06:33:01.558707Z","iopub.execute_input":"2024-07-24T06:33:01.559487Z","iopub.status.idle":"2024-07-24T06:33:01.610689Z","shell.execute_reply.started":"2024-07-24T06:33:01.559454Z","shell.execute_reply":"2024-07-24T06:33:01.609355Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import pipeline\npipe = pipeline(task=\"automatic-speech-recognition\",\n                model='/kaggle/working/whisper-small-dv')\n#pipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")","metadata":{"execution":{"iopub.status.busy":"2024-07-24T06:34:37.376583Z","iopub.execute_input":"2024-07-24T06:34:37.376954Z","iopub.status.idle":"2024-07-24T06:34:41.131669Z","shell.execute_reply.started":"2024-07-24T06:34:37.376923Z","shell.execute_reply":"2024-07-24T06:34:41.130639Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}