{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":73047,"databundleVersionId":8823072,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\n\nimport librosa\nimport librosa.display\n\nimport numpy as np\nimport copy\n\nimport IPython.display as ipd\n\nimport matplotlib.pyplot as plt\n\nimport random\n\nfrom collections import Counter\nfrom torch.utils.data import DataLoader\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torchaudio\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nfrom datasets import DatasetDict\nfrom datasets import Dataset as DS\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer,\n    TrainerCallback,\n    TrainingArguments,\n    TrainerState,\n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline,\n    AdamW\n)\n\nfrom torchmetrics.text import WordErrorRate, CharErrorRate\n\n# !pip freeze > requirements.txt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:43:02.395076Z","iopub.execute_input":"2024-11-20T22:43:02.395475Z","iopub.status.idle":"2024-11-20T22:43:02.401825Z","shell.execute_reply.started":"2024-11-20T22:43:02.395421Z","shell.execute_reply":"2024-11-20T22:43:02.4008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/ben10/ben10'\ntrain_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ntest_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ndata_path = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/train.csv\"\n\ndata = pd.read_csv(data_path)\ndata[\"transcriptions\"] = data[\"transcriptions\"].str.strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:40:59.885829Z","iopub.execute_input":"2024-11-20T22:40:59.88618Z","iopub.status.idle":"2024-11-20T22:41:00.030442Z","shell.execute_reply.started":"2024-11-20T22:40:59.886149Z","shell.execute_reply":"2024-11-20T22:41:00.029384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n        \n        torch.cuda.empty_cache()\n\n        return batch\n\n# data_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:41:01.184083Z","iopub.execute_input":"2024-11-20T22:41:01.184455Z","iopub.status.idle":"2024-11-20T22:41:01.191934Z","shell.execute_reply.started":"2024-11-20T22:41:01.184423Z","shell.execute_reply":"2024-11-20T22:41:01.191047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_dataset(example):\n    audio_path = train_data_dir + example[\"file_name\"]\n    \n    \n    # load the audio using librosa or torch audio (as you wish)\n    audio, sr = librosa.load(audio_path, sr=16_000)\n    \n    example[\"input_features\"] = feature_extractor(audio, sampling_rate=sr).input_features[0]\n    \n    example[\"labels\"] = tokenizer(f\"{example['transcriptions']}\", max_length=448, padding=True, truncation=True).input_ids\n    \n    return example\n\n\ndef filter_inputs(input_audio):\n    \"\"\"filter inputs with zero input length\"\"\"\n    return 0 < len(input_audio)\n\n\ndef filter_labels(input_labels):\n    \"\"\"filter empty label sequences\"\"\"\n    return 0 < len(input_labels)   \n\n\ncer = CharErrorRate()\nwer = WordErrorRate()\n\ndef compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    label_ids[label_ids == -100] = tokenizer.pad_token_id\n\n    pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = tokenizer.batch_decode(label_ids, skip_special_tokens=True)\n\n    wer_res = wer(pred_str, label_str)\n    cer_res = cer(pred_str, label_str)\n    \n    \"\"\"\n        uncomment the next 3 lines if you want to see how the examples look like during eval \n    \"\"\"\n    print(\"WER:\",wer_res,\"| CER:\", cer_res) # to show up during running logs\n    print(\"Pred:\",pred_str[0])\n    print(\"Label:\",label_str[0])\n    \n    return {\"wer\": wer_res, \"cer\": cer_res}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:41:01.991863Z","iopub.execute_input":"2024-11-20T22:41:01.992215Z","iopub.status.idle":"2024-11-20T22:41:02.009095Z","shell.execute_reply.started":"2024-11-20T22:41:01.992186Z","shell.execute_reply":"2024-11-20T22:41:02.00822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = data\ntrain_df, eval_df = train_test_split(train_df, test_size=0.50, shuffle=True)\nlen(train_df), len(eval_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:41:03.462136Z","iopub.execute_input":"2024-11-20T22:41:03.462792Z","iopub.status.idle":"2024-11-20T22:41:03.472988Z","shell.execute_reply.started":"2024-11-20T22:41:03.462757Z","shell.execute_reply":"2024-11-20T22:41:03.472139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef evaluate_model(model, ds_eval, tokenizer, data_collator, batch_size=1, device=\"cuda\"):\n    \"\"\"\n    Evaluate the model on the evaluation dataset and compute the Word Error Rate (WER).\n    \n    Args:\n        model: The pre-trained Whisper model.\n        ds_eval: The evaluation dataset (a Hugging Face Dataset or similar format).\n        tokenizer: The tokenizer used for decoding model predictions.\n        data_collator: The data collator to handle batching and preprocessing.\n        batch_size: The batch size for evaluation.\n        device: The device to use for evaluation (\"cuda\" or \"cpu\").\n    \n    Returns:\n        float: The Word Error Rate (WER) for the evaluation set.\n        list: Predictions generated by the model.\n        list: Ground truth references from the evaluation dataset.\n    \"\"\"\n    # Set up the data loader\n    test_loader = DataLoader(\n        ds_eval,\n        batch_size=batch_size,\n        collate_fn=data_collator,\n    )\n\n    # Place the model in evaluation mode\n    model.eval()\n\n    # Initialize accumulators\n    predictions = []\n    references = []\n\n    # Perform inference\n    for batch in test_loader:\n        # Move inputs to the specified device\n        input_features = batch[\"input_features\"].to(device)\n        \n        # Generate predictions\n        with torch.no_grad():\n            pred_ids = model.generate(input_features)\n        \n        # Decode predictions and references\n        preds = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n        refs = tokenizer.batch_decode(batch[\"labels\"], skip_special_tokens=True)\n        \n        predictions.extend(preds)\n        references.extend(refs)\n\n    # Calculate Word Error Rate (WER)\n    test_wer = wer(references, predictions)\n    \n    # Return results\n    return test_wer, predictions, references\n\n\ndef select_top_samples(eval_subset, top_percent=10):\n    \"\"\"\n    Randomly select the top N% samples from the evaluation subset.\n    \n    Args:\n        eval_subset: The evaluation subset (list, Dataset, or similar iterable).\n        top_percent: The percentage of samples to select (default is 10%).\n\n    Returns:\n        list: Randomly selected top N% samples.\n    \"\"\"\n    # Calculate the number of samples to select\n    num_samples = max(1, int(len(eval_subset) * (top_percent / 100)))\n    \n    # Randomly select samples\n    selected_samples = random.sample(eval_subset, num_samples)\n    \n    return selected_samples\n\ndef replenish_eval_set(eval_subset, top_samples, mother_eval_set, used_samples_pool):\n    \"\"\"\n    Replenish the eval_subset after removing top_samples, ensuring no duplicates,\n    and keep track of all used samples across iterations.\n\n    Args:\n        eval_subset: Current evaluation subset (list of samples or dataset objects).\n        top_samples: Samples to be removed from eval_subset.\n        mother_eval_set: The entire evaluation dataset to draw new samples from.\n        used_samples_pool: A set to keep track of all used sample IDs.\n\n    Returns:\n        Updated eval_subset after replenishment.\n        Updated used_samples_pool with the newly used sample IDs.\n    \"\"\"\n    # Remove top_samples from eval_subset\n    eval_subset = [sample for sample in eval_subset if sample not in top_samples]\n    \n    # Update the used_samples_pool with IDs of the top_samples\n    used_samples_pool.update({id(sample) for sample in top_samples})\n    \n    # Identify available samples in mother_eval_set not already used\n    available_samples = [sample for sample in mother_eval_set if id(sample) not in used_samples_pool]\n    \n    # Ensure enough new samples are available\n    num_new_samples = len(top_samples)\n    if len(available_samples) < num_new_samples:\n        raise ValueError(\"Not enough new samples in the mother evaluation dataset to replenish.\")\n    \n    # Randomly select new samples\n    new_samples = random.sample(available_samples, num_new_samples)\n    \n    # Add new samples to eval_subset\n    eval_subset.extend(new_samples)\n    \n    # Update the used_samples_pool with IDs of the new samples\n    used_samples_pool.update({id(sample) for sample in new_samples})\n    \n    return eval_subset, used_samples_pool\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T22:41:05.378537Z","iopub.execute_input":"2024-11-20T22:41:05.378909Z","iopub.status.idle":"2024-11-20T22:41:05.389959Z","shell.execute_reply.started":"2024-11-20T22:41:05.378878Z","shell.execute_reply":"2024-11-20T22:41:05.389073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main Loop","metadata":{}},{"cell_type":"code","source":"#Setting up base model\nMODEL_NAME = \"openai/whisper-small\"\nbase_model = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, device_map=\"auto\")\nmodel_id = \"whisper-small\"\n\nTASK = \"transcribe\"\nfeature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\ntokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME, language='bn', task=TASK)\nprocessor = WhisperProcessor.from_pretrained(MODEL_NAME, language='bn', task=TASK)\nids = tokenizer.encode(\"\")\ntokenizer.decode(ids)\n\ndata_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)\n\n\n\n\n#Setting up mother dataset\nmother_eval_set = train_df\nmother_train_set = eval_df\nused_samples_pool = set()\n\n\n# Configurations\neval_subset_size = 100  # Size of the evaluation subset\ntrain_subset_size = 100  # Size of the training subset\niterations = 10          # Number of iterations\niteration_logs = []\ntrain_split = DS.from_pandas(mother_train_set[:train_subset_size])\neval_split = DS.from_pandas(mother_eval_set[:eval_subset_size])\n\nfor i in range(iterations):\n    print(f\"Iteration {i + 1}/{iterations}\")\n    model = copy.deepcopy(base_model)\n    model_id = f\"whisper-small/{i}\"\n    #Setting up training and eval data\n    # ben_reg_voice_ds = DatasetDict()\n\n    \n    ds_splits = DatasetDict({\n        'train': train_split,\n        'eval': eval_split\n    })\n    np.object = object\n    ds_splits = ds_splits.map(prepare_dataset, remove_columns=ds_splits.column_names[\"train\"])\n\n    training_args = Seq2SeqTrainingArguments(\n    output_dir=model_id,\n    per_device_train_batch_size=4,\n    per_device_eval_batch_size=4,\n    gradient_accumulation_steps=1,\n    gradient_checkpointing=True,\n    fp16=True,\n    learning_rate=5e-5,\n    weight_decay=1e-2,\n    warmup_steps=100,\n    num_train_epochs=2,\n    evaluation_strategy=\"epoch\", # or \"epochs\"\n    save_strategy=\"epoch\",\n    predict_with_generate=True,\n    generation_max_length=448,\n#     save_steps=2976,\n#     eval_steps=32,\n#     logging_steps=1000,\n    save_total_limit=1,\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    push_to_hub=False,\n    report_to=\"none\",\n    remove_unused_columns=False,\n)\n    \n    model.generation_config.language = \"bn\"\n    model.generation_config.task = \"transcribe\"\n    \n    # model.generation_config.forced_decoder_ids = None\n    model.config.suppress_tokens = [] \n    \n    #Start training\n    optimizer = AdamW(model.parameters(), lr=training_args.learning_rate)\n    \n    trainer = Seq2SeqTrainer(\n        args=training_args,\n        model=model,\n        train_dataset=ds_splits[\"train\"],\n        eval_dataset=ds_splits[\"eval\"],\n        data_collator=data_collator,\n        tokenizer=processor.feature_extractor,\n        compute_metrics=compute_metrics,\n        optimizers=(optimizer, None),\n    #     callbacks=[EarlyStoppingCallback(2, 1.0)]\n    )\n\n        # Evaluate the model\n    test_wer, predictions, references = evaluate_model(\n        model=model,\n        ds_eval=ds_splits[\"eval\"],\n        tokenizer=tokenizer,\n        data_collator=data_collator,\n        batch_size=1,\n        device=\"cuda\" if torch.cuda.is_available() else \"cpu\"\n    )\n    print(f\"WER before training: {test_wer}\")\n    \n    trainer.train()\n    \n    \n    trainer.save_model(training_args.output_dir)\n    processor.save_pretrained(training_args.output_dir)\n\n    # Log training results\n    iteration_logs.append({\n        'iteration': i+1,\n        'wer_before_training': test_wer,\n        'train_loss': trainer.state.log_history[-1]['loss'] if 'loss' in trainer.state.log_history[-1] else None,\n    })\n    \n    out_logs = pd.DataFrame(iteration_logs)\n    out_logs.to_csv(f\"logs_train_{i}.csv\",index=False)\n\n    # Evaluate the model\n    test_wer, predictions, references = evaluate_model(\n        model=model,\n        ds_eval=ds_splits[\"eval\"],\n        tokenizer=tokenizer,\n        data_collator=data_collator,\n        batch_size=1,\n        device=\"cuda\" if torch.cuda.is_available() else \"cpu\"\n    )\n    print(f\"WER after training: {test_wer}\")\n    iteration_logs[-1]['wer_after_training'] = test_wer\n    \n    # Remove the top 10% samples and replenish the eval_subset\n    top_10_percent_samples = select_top_samples(eval_split, top_percent=10)\n    print(f\"Selected {len(top_10_percent_samples)} top-performing samples.\")\n    \n    train_split = train_split.add_batch(top_10_percent_samples)\n\n    # Replenish the eval_subset with new samples from the mother dataset\n    eval_split, used = replenish_eval_set(\n    eval_split,\n    top_10_percent_samples,\n    mother_eval_set,\n    used_samples_pool\n    )\n    used_samples_pool.update(used)\n        \n    iteration_logs[-1]['num_top_10_percent_samples'] = len(top_10_percent_samples)\n    iteration_logs[-1]['new_eval_subset_size'] = len(eval_split)\n    \n    print(f\"New eval subset size: {len(eval_split)}\")\n    print(\"-\" * 50)\n\nprint(\"Process complete.\")\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}