{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":8212787,"sourceType":"datasetVersion","datasetId":4867389}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\n\nimport librosa\nimport librosa.display\n\nimport numpy as np\n\nimport IPython.display as ipd\n\nimport matplotlib.pyplot as plt\n\nimport random\n\nfrom collections import Counter\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torchaudio\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nfrom datasets import DatasetDict\nfrom datasets import Dataset as DS\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer,\n    TrainerCallback,\n    TrainingArguments,\n    TrainerState,\n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline\n)\n\nfrom torchmetrics.text import WordErrorRate, CharErrorRate","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-24T07:30:08.543626Z","iopub.execute_input":"2024-04-24T07:30:08.54405Z","iopub.status.idle":"2024-04-24T07:30:08.552216Z","shell.execute_reply.started":"2024-04-24T07:30:08.544021Z","shell.execute_reply":"2024-04-24T07:30:08.550577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/ben10/ben10'\ntrain_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ntest_data_dir = f\"{BASE_DIR}/16_kHz_valid_audio/\"\ndata_path = f\"{BASE_DIR}/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:08.554028Z","iopub.execute_input":"2024-04-24T07:30:08.554349Z","iopub.status.idle":"2024-04-24T07:30:08.583823Z","shell.execute_reply.started":"2024-04-24T07:30:08.554321Z","shell.execute_reply":"2024-04-24T07:30:08.5829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split2path = {\n    \"train\": train_data_dir,\n    \"test\": test_data_dir,\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:08.584978Z","iopub.execute_input":"2024-04-24T07:30:08.585299Z","iopub.status.idle":"2024-04-24T07:30:08.594022Z","shell.execute_reply.started":"2024-04-24T07:30:08.585272Z","shell.execute_reply":"2024-04-24T07:30:08.593183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(data_path)\ndata.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:08.596292Z","iopub.execute_input":"2024-04-24T07:30:08.5966Z","iopub.status.idle":"2024-04-24T07:30:08.753691Z","shell.execute_reply.started":"2024-04-24T07:30:08.596573Z","shell.execute_reply":"2024-04-24T07:30:08.752711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_split(filename):\n    filename_ = filename.split(\"_\")\n    split = filename_[0]\n    return split\n\ndef extract_district(filename):\n    filename_ = filename.split(\" \")[0]\n    district = filename_.split(\"_\")[1]\n    return district\n\ndef beautify_dataset(data):\n    splits = []\n    districts = []\n    newpaths = []\n    transcripts = []\n    \n    for i in range(len(data)):\n        filename, transcript = data.iloc[i]\n        split = extract_split(filename)\n        district = extract_district(filename)\n        dir_path = split2path[split]\n        composed_path = f\"{dir_path}{filename}\"\n        \n        if os.path.exists(composed_path) == False:\n            print(f\"{composed_path} does not exist.\")\n            continue\n        \n        # replace any newline characters\n        transcript = transcript.replace(\"\\n\", \" \")\n        transcript = \" \".join(transcript.split())\n        \n        splits.append(split)\n        districts.append(district)\n        newpaths.append(composed_path)\n        transcripts.append(transcript)\n    \n    data['file_path'] = newpaths\n    data['district'] = districts\n    data['split'] = splits\n    data['transcripts'] = transcripts\n    \n#     data.drop(columns=['file_name'], inplace=True)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:08.755327Z","iopub.execute_input":"2024-04-24T07:30:08.755665Z","iopub.status.idle":"2024-04-24T07:30:08.765074Z","shell.execute_reply.started":"2024-04-24T07:30:08.755634Z","shell.execute_reply":"2024-04-24T07:30:08.763981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = beautify_dataset(data)\ndata.sample(20)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:08.766345Z","iopub.execute_input":"2024-04-24T07:30:08.766933Z","iopub.status.idle":"2024-04-24T07:30:16.158123Z","shell.execute_reply.started":"2024-04-24T07:30:08.7669Z","shell.execute_reply":"2024-04-24T07:30:16.157174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"<>\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.159275Z","iopub.execute_input":"2024-04-24T07:30:16.159607Z","iopub.status.idle":"2024-04-24T07:30:16.174589Z","shell.execute_reply.started":"2024-04-24T07:30:16.159581Z","shell.execute_reply":"2024-04-24T07:30:16.173646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.175897Z","iopub.execute_input":"2024-04-24T07:30:16.17624Z","iopub.status.idle":"2024-04-24T07:30:16.191342Z","shell.execute_reply.started":"2024-04-24T07:30:16.176211Z","shell.execute_reply":"2024-04-24T07:30:16.190453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"..\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.194096Z","iopub.execute_input":"2024-04-24T07:30:16.194403Z","iopub.status.idle":"2024-04-24T07:30:16.209119Z","shell.execute_reply.started":"2024-04-24T07:30:16.19438Z","shell.execute_reply":"2024-04-24T07:30:16.208175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTE:** Think of how you want use the existing models/your finetuned model to replace these examples.... For now let's just handle them.","metadata":{}},{"cell_type":"code","source":"# print(list(data[data['transcripts'] == ''].index))\ndata.drop(data[data['transcripts'] == ''].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '<>'].index))\ndata.drop(data[data['transcripts'] == \"<>\"].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '..'].index))\ndata.drop(data[data['transcripts'] == \"..\"].index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.210431Z","iopub.execute_input":"2024-04-24T07:30:16.210748Z","iopub.status.idle":"2024-04-24T07:30:16.234613Z","shell.execute_reply.started":"2024-04-24T07:30:16.210719Z","shell.execute_reply":"2024-04-24T07:30:16.233731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"] = data[\"transcripts\"].str.strip()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.2357Z","iopub.execute_input":"2024-04-24T07:30:16.23594Z","iopub.status.idle":"2024-04-24T07:30:16.244995Z","shell.execute_reply.started":"2024-04-24T07:30:16.235919Z","shell.execute_reply":"2024-04-24T07:30:16.244167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TASK = \"transcribe\"\nMODEL_NAME = \"openai/whisper-medium\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.246185Z","iopub.execute_input":"2024-04-24T07:30:16.246531Z","iopub.status.idle":"2024-04-24T07:30:16.254639Z","shell.execute_reply.started":"2024-04-24T07:30:16.246502Z","shell.execute_reply":"2024-04-24T07:30:16.253781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\ntokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME, language='bn', task=TASK)\nprocessor = WhisperProcessor.from_pretrained(MODEL_NAME, language='bn', task=TASK)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:16.255707Z","iopub.execute_input":"2024-04-24T07:30:16.256006Z","iopub.status.idle":"2024-04-24T07:30:22.012627Z","shell.execute_reply.started":"2024-04-24T07:30:16.255968Z","shell.execute_reply":"2024-04-24T07:30:22.011849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = tokenizer.encode(\"\")\nids","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.01369Z","iopub.execute_input":"2024-04-24T07:30:22.013943Z","iopub.status.idle":"2024-04-24T07:30:22.019852Z","shell.execute_reply.started":"2024-04-24T07:30:22.013921Z","shell.execute_reply":"2024-04-24T07:30:22.019044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.decode(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.021032Z","iopub.execute_input":"2024-04-24T07:30:22.021368Z","iopub.status.idle":"2024-04-24T07:30:22.034991Z","shell.execute_reply.started":"2024-04-24T07:30:22.021338Z","shell.execute_reply":"2024-04-24T07:30:22.0342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n        \n        torch.cuda.empty_cache()\n\n        return batch","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.036752Z","iopub.execute_input":"2024-04-24T07:30:22.037023Z","iopub.status.idle":"2024-04-24T07:30:22.045993Z","shell.execute_reply.started":"2024-04-24T07:30:22.037Z","shell.execute_reply":"2024-04-24T07:30:22.045286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.047119Z","iopub.execute_input":"2024-04-24T07:30:22.047422Z","iopub.status.idle":"2024-04-24T07:30:22.06062Z","shell.execute_reply.started":"2024-04-24T07:30:22.047397Z","shell.execute_reply":"2024-04-24T07:30:22.059765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(example):\n    audio_path = example[\"file_path\"]\n    \n    # load the audio using librosa or torch audio (as you wish)\n    audio, sr = librosa.load(audio_path, sr=16_000)\n    \n    example[\"input_features\"] = feature_extractor(audio, sampling_rate=sr).input_features[0]\n    \n    example[\"labels\"] = tokenizer(f\"{example['transcripts']}\", max_length=448, padding=True, truncation=True).input_ids\n    \n    return example\n\n\ndef filter_inputs(input_audio):\n    \"\"\"filter inputs with zero input length\"\"\"\n    return 0 < len(input_audio)\n\n\ndef filter_labels(input_labels):\n    \"\"\"filter empty label sequences\"\"\"\n    return 0 < len(input_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.06168Z","iopub.execute_input":"2024-04-24T07:30:22.061938Z","iopub.status.idle":"2024-04-24T07:30:22.069869Z","shell.execute_reply.started":"2024-04-24T07:30:22.061916Z","shell.execute_reply":"2024-04-24T07:30:22.069158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = data[data[\"split\"] == \"train\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.070961Z","iopub.execute_input":"2024-04-24T07:30:22.071315Z","iopub.status.idle":"2024-04-24T07:30:22.094356Z","shell.execute_reply.started":"2024-04-24T07:30:22.071286Z","shell.execute_reply":"2024-04-24T07:30:22.093613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    adjust test size accordingly.\n\"\"\"\ntrain_df, eval_df = train_test_split(train_df, test_size=0.1, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.095609Z","iopub.execute_input":"2024-04-24T07:30:22.095893Z","iopub.status.idle":"2024-04-24T07:30:22.103638Z","shell.execute_reply.started":"2024-04-24T07:30:22.095869Z","shell.execute_reply":"2024-04-24T07:30:22.102849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df), len(eval_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.104707Z","iopub.execute_input":"2024-04-24T07:30:22.105057Z","iopub.status.idle":"2024-04-24T07:30:22.112374Z","shell.execute_reply.started":"2024-04-24T07:30:22.105013Z","shell.execute_reply":"2024-04-24T07:30:22.111612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ben_reg_voice_ds = DatasetDict()\n\ntrain_split = DS.from_pandas(train_df)\neval_split = DS.from_pandas(eval_df)\n\nds_splits = DatasetDict({\n    'train': train_split,\n    'eval': eval_split\n})","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.113414Z","iopub.execute_input":"2024-04-24T07:30:22.113698Z","iopub.status.idle":"2024-04-24T07:30:22.228362Z","shell.execute_reply.started":"2024-04-24T07:30:22.113675Z","shell.execute_reply":"2024-04-24T07:30:22.227622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.remove_columns([\"split\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.2334Z","iopub.execute_input":"2024-04-24T07:30:22.233674Z","iopub.status.idle":"2024-04-24T07:30:22.240919Z","shell.execute_reply.started":"2024-04-24T07:30:22.233651Z","shell.execute_reply":"2024-04-24T07:30:22.24004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ds_splits)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.241917Z","iopub.execute_input":"2024-04-24T07:30:22.242231Z","iopub.status.idle":"2024-04-24T07:30:22.25261Z","shell.execute_reply.started":"2024-04-24T07:30:22.242207Z","shell.execute_reply":"2024-04-24T07:30:22.251724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.object = object","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.253516Z","iopub.execute_input":"2024-04-24T07:30:22.253791Z","iopub.status.idle":"2024-04-24T07:30:22.261575Z","shell.execute_reply.started":"2024-04-24T07:30:22.253769Z","shell.execute_reply":"2024-04-24T07:30:22.260704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('multithreadding area started -------------')","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.262684Z","iopub.execute_input":"2024-04-24T07:30:22.262938Z","iopub.status.idle":"2024-04-24T07:30:22.271181Z","shell.execute_reply.started":"2024-04-24T07:30:22.262916Z","shell.execute_reply":"2024-04-24T07:30:22.270447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.map(prepare_dataset, remove_columns=ds_splits.column_names[\"train\"],\n                          num_proc=None # open for multithreadding\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:30:22.272207Z","iopub.execute_input":"2024-04-24T07:30:22.272522Z","iopub.status.idle":"2024-04-24T07:37:11.018126Z","shell.execute_reply.started":"2024-04-24T07:30:22.272493Z","shell.execute_reply":"2024-04-24T07:37:11.0172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ds_splits = ds_splits.filter(filter_inputs, input_columns=[\"input_features\"])\n# ds_splits = ds_splits.filter(filter_labels, input_columns=[\"labels\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.019627Z","iopub.execute_input":"2024-04-24T07:37:11.020006Z","iopub.status.idle":"2024-04-24T07:37:11.024472Z","shell.execute_reply.started":"2024-04-24T07:37:11.019964Z","shell.execute_reply":"2024-04-24T07:37:11.023598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ds_splits[\"train\"]), len(ds_splits[\"eval\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.025621Z","iopub.execute_input":"2024-04-24T07:37:11.025924Z","iopub.status.idle":"2024-04-24T07:37:11.038733Z","shell.execute_reply.started":"2024-04-24T07:37:11.0259Z","shell.execute_reply":"2024-04-24T07:37:11.037789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cer = CharErrorRate()\nwer = WordErrorRate()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.039897Z","iopub.execute_input":"2024-04-24T07:37:11.0402Z","iopub.status.idle":"2024-04-24T07:37:11.050452Z","shell.execute_reply.started":"2024-04-24T07:37:11.040176Z","shell.execute_reply":"2024-04-24T07:37:11.049711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    label_ids[label_ids == -100] = tokenizer.pad_token_id\n\n    pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = tokenizer.batch_decode(label_ids, skip_special_tokens=True)\n\n    wer_res = wer(pred_str, label_str)\n    cer_res = cer(pred_str, label_str)\n    \n    \"\"\"\n        uncomment the next 3 lines if you want to see how the examples look like during eval \n    \"\"\"\n    print(\"WER:\",wer_res,\"| CER:\", cer_res) # to show up during running logs\n    print(\"Pred:\",pred_str[0])\n    print(\"Label:\",label_str[0])\n    \n    return {\"wer\": wer_res, \"cer\": cer_res}","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.051421Z","iopub.execute_input":"2024-04-24T07:37:11.051728Z","iopub.status.idle":"2024-04-24T07:37:11.058685Z","shell.execute_reply.started":"2024-04-24T07:37:11.051694Z","shell.execute_reply":"2024-04-24T07:37:11.057824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training started ...................')","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.059638Z","iopub.execute_input":"2024-04-24T07:37:11.059917Z","iopub.status.idle":"2024-04-24T07:37:11.072327Z","shell.execute_reply.started":"2024-04-24T07:37:11.059876Z","shell.execute_reply":"2024-04-24T07:37:11.071379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, device_map=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:11.073362Z","iopub.execute_input":"2024-04-24T07:37:11.073653Z","iopub.status.idle":"2024-04-24T07:37:30.150249Z","shell.execute_reply.started":"2024-04-24T07:37:11.073628Z","shell.execute_reply":"2024-04-24T07:37:30.149329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_id = \"whisper-reg-ben\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:30.151436Z","iopub.execute_input":"2024-04-24T07:37:30.151744Z","iopub.status.idle":"2024-04-24T07:37:30.157918Z","shell.execute_reply.started":"2024-04-24T07:37:30.151718Z","shell.execute_reply":"2024-04-24T07:37:30.157055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = Seq2SeqTrainingArguments(\n    output_dir=model_id,\n    per_device_train_batch_size=8,\n    per_device_eval_batch_size=8,\n    gradient_accumulation_steps=1,\n    gradient_checkpointing=True,\n    fp16=True,\n    learning_rate=1e-4,\n    weight_decay=5e-3,\n    warmup_steps=100,\n    num_train_epochs=1,\n    evaluation_strategy=\"steps\", # or \"epochs\"\n    predict_with_generate=True,\n#     generation_max_length=448,\n    save_steps=1000,\n    eval_steps=1000,\n    logging_steps=1000,\n    save_total_limit=1,\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    push_to_hub=False,\n    report_to=\"none\",\n    remove_unused_columns=False,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:30.158884Z","iopub.execute_input":"2024-04-24T07:37:30.159161Z","iopub.status.idle":"2024-04-24T07:37:31.556467Z","shell.execute_reply.started":"2024-04-24T07:37:30.159117Z","shell.execute_reply":"2024-04-24T07:37:31.555672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.generation_config.language = \"bn\"\nmodel.generation_config.task = \"transcribe\"\n\nmodel.generation_config.forced_decoder_ids = None\nmodel.config.suppress_tokens = [] # added later","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:31.557625Z","iopub.execute_input":"2024-04-24T07:37:31.558051Z","iopub.status.idle":"2024-04-24T07:37:31.563188Z","shell.execute_reply.started":"2024-04-24T07:37:31.558017Z","shell.execute_reply":"2024-04-24T07:37:31.562366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=ds_splits[\"train\"],\n    eval_dataset=ds_splits[\"eval\"],\n    data_collator=data_collator,\n    tokenizer=processor.feature_extractor,\n    compute_metrics=compute_metrics,\n#     callbacks=[EarlyStoppingCallback(2, 1.0)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:31.564546Z","iopub.execute_input":"2024-04-24T07:37:31.565353Z","iopub.status.idle":"2024-04-24T07:37:31.594126Z","shell.execute_reply.started":"2024-04-24T07:37:31.565327Z","shell.execute_reply":"2024-04-24T07:37:31.593297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# trainer.train()\n\n# # to use the high-level pipeline, ensure both the processor outputs and model outputs exist in the same dir\n# trainer.save_model(training_args.output_dir)\n# processor.save_pretrained(training_args.output_dir)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:31.595425Z","iopub.execute_input":"2024-04-24T07:37:31.595728Z","iopub.status.idle":"2024-04-24T07:37:31.601442Z","shell.execute_reply.started":"2024-04-24T07:37:31.595702Z","shell.execute_reply":"2024-04-24T07:37:31.600669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# out_logs = pd.DataFrame(trainer.state.log_history)\n# out_logs.to_csv(\"logs.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:31.602485Z","iopub.execute_input":"2024-04-24T07:37:31.602752Z","iopub.status.idle":"2024-04-24T07:37:31.610046Z","shell.execute_reply.started":"2024-04-24T07:37:31.602728Z","shell.execute_reply":"2024-04-24T07:37:31.609186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel ds_splits\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:31.61105Z","iopub.execute_input":"2024-04-24T07:37:31.611326Z","iopub.status.idle":"2024-04-24T07:37:32.045191Z","shell.execute_reply.started":"2024-04-24T07:37:31.611302Z","shell.execute_reply":"2024-04-24T07:37:32.044272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:32.046516Z","iopub.execute_input":"2024-04-24T07:37:32.046791Z","iopub.status.idle":"2024-04-24T07:37:32.054872Z","shell.execute_reply.started":"2024-04-24T07:37:32.046767Z","shell.execute_reply":"2024-04-24T07:37:32.054009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Device:\", device)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:32.055835Z","iopub.execute_input":"2024-04-24T07:37:32.056084Z","iopub.status.idle":"2024-04-24T07:37:32.066882Z","shell.execute_reply.started":"2024-04-24T07:37:32.056062Z","shell.execute_reply":"2024-04-24T07:37:32.066025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe = pipeline(\n    \"automatic-speech-recognition\",\n    model='/kaggle/input/tem-dataset-for-whisper/whisper-reg-ben',\n    chunk_length_s=24.1,\n    device=0,\n    batch_size = 8,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:32.067984Z","iopub.execute_input":"2024-04-24T07:37:32.06831Z","iopub.status.idle":"2024-04-24T07:37:58.946978Z","shell.execute_reply.started":"2024-04-24T07:37:32.068285Z","shell.execute_reply":"2024-04-24T07:37:58.946159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pretty_sort(filename):\n    name, number_str = filename.split(\" (\")\n    number = int(number_str.split(\")\")[0])\n    return name, number","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:58.948076Z","iopub.execute_input":"2024-04-24T07:37:58.948395Z","iopub.status.idle":"2024-04-24T07:37:58.953331Z","shell.execute_reply.started":"2024-04-24T07:37:58.94837Z","shell.execute_reply":"2024-04-24T07:37:58.952343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\npreds = []","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:58.954505Z","iopub.execute_input":"2024-04-24T07:37:58.954842Z","iopub.status.idle":"2024-04-24T07:37:58.964643Z","shell.execute_reply.started":"2024-04-24T07:37:58.954817Z","shell.execute_reply":"2024-04-24T07:37:58.963809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for root, dirs, files in os.walk(\"/kaggle/input/ben10/ben10/16_kHz_valid_audio\"):\n    files = sorted(files, key=pretty_sort)\n    \n#     print(files.index(\"valid_sandwip (1).wav\"))\n#     print(files.index(\"valid_sandwip (132).wav\"))\n    \n#     put swandip first\n    shift = files[1070 : 1202]\n    \n    files = shift + files[:1070] + files[1202:]\n    ids = files.copy()\n    \n    for file in files:\n        composed_path = f\"{test_data_dir}{file}\"\n        audio, sr = librosa.load(composed_path, sr=16_000)\n        text = pipe(audio)[\"text\"]\n        preds.append(text)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T07:37:58.96575Z","iopub.execute_input":"2024-04-24T07:37:58.966078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[\"id\"] = ids\nsub_df[\"sentence\"] = preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}