{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Fine-tuning whisper with LoRA\n\nWith this notebook an efficient fine tuning of the openai whisper model can be performed using LoRA. To use it, the respective api tokens must be added.","metadata":{}},{"cell_type":"markdown","source":"# Install dependencies","metadata":{}},{"cell_type":"code","source":"!pip install peft transformers  accelerate jiwer bitsandbytes -q # datasets","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:36:29.511214Z","iopub.execute_input":"2023-09-16T21:36:29.511481Z","iopub.status.idle":"2023-09-16T21:36:47.67599Z","shell.execute_reply.started":"2023-09-16T21:36:29.511437Z","shell.execute_reply":"2023-09-16T21:36:47.674766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U -q datasets","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:36:47.679386Z","iopub.execute_input":"2023-09-16T21:36:47.680025Z","iopub.status.idle":"2023-09-16T21:37:01.667147Z","shell.execute_reply.started":"2023-09-16T21:36:47.679983Z","shell.execute_reply":"2023-09-16T21:37:01.665831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \n\nos.environ[\"WANDB_API_KEY\"] = \"\"","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:37:01.669693Z","iopub.execute_input":"2023-09-16T21:37:01.670147Z","iopub.status.idle":"2023-09-16T21:37:01.679952Z","shell.execute_reply.started":"2023-09-16T21:37:01.670094Z","shell.execute_reply":"2023-09-16T21:37:01.678805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from huggingface_hub import notebook_login\n\nnotebook_login()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:37:01.683624Z","iopub.execute_input":"2023-09-16T21:37:01.68463Z","iopub.status.idle":"2023-09-16T21:37:01.970499Z","shell.execute_reply.started":"2023-09-16T21:37:01.68458Z","shell.execute_reply":"2023-09-16T21:37:01.969601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import load_dataset\nfrom datasets import load_dataset, DatasetDict\nfrom pathlib import Path\nimport torch \nimport os","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:37:01.971869Z","iopub.execute_input":"2023-09-16T21:37:01.972844Z","iopub.status.idle":"2023-09-16T21:37:06.04479Z","shell.execute_reply.started":"2023-09-16T21:37:01.972809Z","shell.execute_reply":"2023-09-16T21:37:06.043414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data\n\nThe data used comes from an already preproccessed version of a subset of Common Voice 13 dataset (bengali only).","metadata":{}},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = DATA / \"test_mp3s\"\nTRAIN_PARQUET = INPUT / 'preprocessed-common-voice-13-bengali'\n\nSAMPLING_RATE = 16_000","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:37:06.046555Z","iopub.execute_input":"2023-09-16T21:37:06.047307Z","iopub.status.idle":"2023-09-16T21:37:06.053516Z","shell.execute_reply.started":"2023-09-16T21:37:06.047266Z","shell.execute_reply":"2023-09-16T21:37:06.05201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorized_datasets = DatasetDict()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:37:06.055402Z","iopub.execute_input":"2023-09-16T21:37:06.056046Z","iopub.status.idle":"2023-09-16T21:37:06.07158Z","shell.execute_reply.started":"2023-09-16T21:37:06.056011Z","shell.execute_reply":"2023-09-16T21:37:06.070647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = list(map(str, TRAIN_PARQUET.glob(\"train*.parquet\")))\nvectorized_datasets[\"train\"] = load_dataset(\n            \"parquet\", data_files=train_files, split=\"train\"\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-16T21:37:06.073108Z","iopub.execute_input":"2023-09-16T21:37:06.073442Z","iopub.status.idle":"2023-09-16T21:38:32.179857Z","shell.execute_reply.started":"2023-09-16T21:37:06.07341Z","shell.execute_reply":"2023-09-16T21:38:32.178549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_files = list(map(str, TRAIN_PARQUET.glob(\"test*.parquet\")))\nvectorized_datasets[\"eval\"] = load_dataset(\n    \"parquet\", data_files=eval_files, split=\"train\"\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:38:32.185962Z","iopub.execute_input":"2023-09-16T21:38:32.18881Z","iopub.status.idle":"2023-09-16T21:38:40.793196Z","shell.execute_reply.started":"2023-09-16T21:38:32.188766Z","shell.execute_reply":"2023-09-16T21:38:40.791741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine-Tuning","metadata":{}},{"cell_type":"code","source":"model_name_or_path = 'openai/whisper-small' # \"openai/whisper-large-v2\"\nlanguage = \"Bengali\"\nlanguage_abbr = \"bn\"\ntask = \"transcribe\"","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:38:40.801308Z","iopub.execute_input":"2023-09-16T21:38:40.801773Z","iopub.status.idle":"2023-09-16T21:38:40.810678Z","shell.execute_reply.started":"2023-09-16T21:38:40.801729Z","shell.execute_reply":"2023-09-16T21:38:40.809685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoFeatureExtractor, AutoTokenizer, AutoProcessor\n\nfeature_extractor = AutoFeatureExtractor.from_pretrained(model_name_or_path)\ntokenizer = AutoTokenizer.from_pretrained(model_name_or_path, language=language, task=task)\nprocessor = AutoProcessor.from_pretrained(model_name_or_path, language=language, task=task)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:38:40.81441Z","iopub.execute_input":"2023-09-16T21:38:40.815174Z","iopub.status.idle":"2023-09-16T21:38:53.285006Z","shell.execute_reply.started":"2023-09-16T21:38:40.815133Z","shell.execute_reply":"2023-09-16T21:38:53.284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\n\n\n@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n\n        return batch\n\n\ndata_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:38:53.286788Z","iopub.execute_input":"2023-09-16T21:38:53.287554Z","iopub.status.idle":"2023-09-16T21:38:53.297743Z","shell.execute_reply.started":"2023-09-16T21:38:53.287516Z","shell.execute_reply":"2023-09-16T21:38:53.296629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModelForSpeechSeq2Seq\n\nmodel = AutoModelForSpeechSeq2Seq.from_pretrained(model_name_or_path, load_in_8bit=True, device_map=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:38:53.299337Z","iopub.execute_input":"2023-09-16T21:38:53.299956Z","iopub.status.idle":"2023-09-16T21:39:06.79691Z","shell.execute_reply.started":"2023-09-16T21:38:53.299921Z","shell.execute_reply":"2023-09-16T21:39:06.795138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.config.forced_decoder_ids = None\nmodel.config.suppress_tokens = []","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:06.80444Z","iopub.execute_input":"2023-09-16T21:39:06.804738Z","iopub.status.idle":"2023-09-16T21:39:06.812692Z","shell.execute_reply.started":"2023-09-16T21:39:06.804711Z","shell.execute_reply":"2023-09-16T21:39:06.811684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from peft import prepare_model_for_int8_training\n\nmodel = prepare_model_for_int8_training(model)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:06.814038Z","iopub.execute_input":"2023-09-16T21:39:06.8191Z","iopub.status.idle":"2023-09-16T21:39:09.552843Z","shell.execute_reply.started":"2023-09-16T21:39:06.819042Z","shell.execute_reply":"2023-09-16T21:39:09.550703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LoRA\n\nAbout LoRA parameters:  \nThe weight matrix is scaled by lora_alpha/r, and a higher lora_alpha value assigns more weight to the LoRA activations. For performance, we recommend setting bias to None first, and then lora_only, before trying all.","metadata":{}},{"cell_type":"code","source":"from peft import LoraConfig, PeftModel, LoraModel, LoraConfig, get_peft_model\n\nconfig = LoraConfig(r=32, lora_alpha=64, target_modules=[\"q_proj\", \"v_proj\"], lora_dropout=0.05, bias=\"none\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:09.554498Z","iopub.execute_input":"2023-09-16T21:39:09.554866Z","iopub.status.idle":"2023-09-16T21:39:09.561578Z","shell.execute_reply.started":"2023-09-16T21:39:09.554832Z","shell.execute_reply":"2023-09-16T21:39:09.560212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_peft_model(model, config)\nmodel.print_trainable_parameters()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:09.563133Z","iopub.execute_input":"2023-09-16T21:39:09.563611Z","iopub.status.idle":"2023-09-16T21:39:10.235842Z","shell.execute_reply.started":"2023-09-16T21:39:09.563577Z","shell.execute_reply":"2023-09-16T21:39:10.234764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Seq2SeqTrainingArguments\n\ntraining_args = Seq2SeqTrainingArguments(\n    output_dir=\"/kaggle/working/david/int8-whisper-small-asr-bengali\",\n    per_device_train_batch_size=8,\n    gradient_accumulation_steps=1,\n    learning_rate=1e-3,\n    warmup_steps=50,\n    num_train_epochs=5,\n    evaluation_strategy=\"epoch\",\n    fp16=True,\n    per_device_eval_batch_size=8,\n    generation_max_length=128,\n    logging_steps=25,\n    remove_unused_columns=False,\n    label_names=[\"labels\"],\n    push_to_hub=False,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:45.571939Z","iopub.execute_input":"2023-09-16T21:39:45.572674Z","iopub.status.idle":"2023-09-16T21:39:45.589349Z","shell.execute_reply.started":"2023-09-16T21:39:45.572639Z","shell.execute_reply":"2023-09-16T21:39:45.588282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers.trainer_utils import PREFIX_CHECKPOINT_DIR\nfrom transformers import Seq2SeqTrainer, TrainerCallback, Seq2SeqTrainingArguments, TrainerState, TrainerControl\n\nclass SavePeftModelCallback(TrainerCallback):\n    def on_save(\n        self,\n        args: Seq2SeqTrainingArguments, # WARNING: lo he cambiado yo, puede que si explota se apor esto\n        state: TrainerState,\n        control: TrainerControl,\n        **kwargs,\n    ):\n        checkpoint_folder = os.path.join(args.output_dir, f\"{PREFIX_CHECKPOINT_DIR}-{state.global_step}\")\n\n        peft_model_path = os.path.join(checkpoint_folder, \"adapter_model\")\n        kwargs[\"model\"].save_pretrained(peft_model_path)\n\n#         pytorch_model_path = os.path.join(checkpoint_folder, \"pytorch_model.bin\")\n#         if os.path.exists(pytorch_model_path):\n#             os.remove(pytorch_model_path)\n        return control","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:45.864959Z","iopub.execute_input":"2023-09-16T21:39:45.865301Z","iopub.status.idle":"2023-09-16T21:39:46.073991Z","shell.execute_reply.started":"2023-09-16T21:39:45.865275Z","shell.execute_reply":"2023-09-16T21:39:46.073016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Seq2SeqTrainer, TrainerCallback, Seq2SeqTrainingArguments, TrainerState, TrainerControl\n\ntrainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=vectorized_datasets[\"train\"],\n    eval_dataset=vectorized_datasets[\"eval\"],\n    data_collator=data_collator,\n    tokenizer=processor.feature_extractor,\n    callbacks=[SavePeftModelCallback],\n\n)\nmodel.config.use_cache = False\n","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:46.404532Z","iopub.execute_input":"2023-09-16T21:39:46.405269Z","iopub.status.idle":"2023-09-16T21:39:46.41702Z","shell.execute_reply.started":"2023-09-16T21:39:46.405226Z","shell.execute_reply":"2023-09-16T21:39:46.416151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  trainer.train()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:48.626871Z","iopub.execute_input":"2023-09-16T21:39:48.627568Z","iopub.status.idle":"2023-09-16T21:39:48.632779Z","shell.execute_reply.started":"2023-09-16T21:39:48.627517Z","shell.execute_reply":"2023-09-16T21:39:48.631722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with torch.autocast(\"cuda\"):\n    trainer.train()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:48.921084Z","iopub.execute_input":"2023-09-16T21:39:48.921738Z","iopub.status.idle":"2023-09-16T21:42:07.215997Z","shell.execute_reply.started":"2023-09-16T21:39:48.921709Z","shell.execute_reply":"2023-09-16T21:42:07.212719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"HUGGINGFACE_TOKEN\"] = \"\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.push_to_hub(\"davidramos/int8-whisper-small-asr-bengali-20kdata\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:11.1623Z","iopub.status.idle":"2023-09-16T21:39:11.163267Z","shell.execute_reply.started":"2023-09-16T21:39:11.162989Z","shell.execute_reply":"2023-09-16T21:39:11.163017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.save_model()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:11.164672Z","iopub.status.idle":"2023-09-16T21:39:11.165176Z","shell.execute_reply.started":"2023-09-16T21:39:11.164902Z","shell.execute_reply":"2023-09-16T21:39:11.164925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate the model","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import DataLoader\nfrom tqdm import tqdm\nimport numpy as np\nimport gc\n\neval_dataloader = DataLoader(vectorized_datasets[\"eval\"], batch_size=8, collate_fn=data_collator)\n\nmodel.eval()\nfor step, batch in enumerate(tqdm(eval_dataloader)):\n    with torch.cuda.amp.autocast():\n        with torch.no_grad():\n            generated_tokens = (\n                model.generate(\n                    input_features=batch[\"input_features\"].to(\"cuda\"),\n                    decoder_input_ids=batch[\"labels\"][:, :4].to(\"cuda\"),\n                    max_new_tokens=255,\n                )\n                .cpu()\n                .numpy()\n            )\n            labels = batch[\"labels\"].cpu().numpy()\n            labels = np.where(labels != -100, labels, tokenizer.pad_token_id)\n            decoded_preds = tokenizer.batch_decode(generated_tokens, skip_special_tokens=True)\n            decoded_labels = tokenizer.batch_decode(labels, skip_special_tokens=True)\n            metric.add_batch(\n                predictions=decoded_preds,\n                references=decoded_labels,\n            )\n    del generated_tokens, labels, batch\n    gc.collect()\nwer = 100 * metric.compute()\nprint(f\"{wer=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T21:39:11.167148Z","iopub.status.idle":"2023-09-16T21:39:11.168417Z","shell.execute_reply.started":"2023-09-16T21:39:11.168166Z","shell.execute_reply":"2023-09-16T21:39:11.168191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}