{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":73047,"databundleVersionId":8140249,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\n\nimport librosa\nimport librosa.display\n\nimport numpy as np\n\nimport IPython.display as ipd\n\nimport matplotlib.pyplot as plt\n\nimport random\n\nfrom collections import Counter\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torchaudio\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nfrom datasets import DatasetDict\nfrom datasets import Dataset as DS\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer,\n    TrainerCallback,\n    TrainingArguments,\n    TrainerState,\n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline\n)\n\nfrom torchmetrics.text import WordErrorRate, CharErrorRate","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-04T11:25:08.949983Z","iopub.execute_input":"2024-04-04T11:25:08.950356Z","iopub.status.idle":"2024-04-04T11:25:35.814431Z","shell.execute_reply.started":"2024-04-04T11:25:08.950327Z","shell.execute_reply":"2024-04-04T11:25:35.813435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/ben10/ben10'\ntrain_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ntest_data_dir = f\"{BASE_DIR}/16_kHz_valid_audio/\"\ndata_path = f\"{BASE_DIR}/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:35.816065Z","iopub.execute_input":"2024-04-04T11:25:35.816678Z","iopub.status.idle":"2024-04-04T11:25:35.821662Z","shell.execute_reply.started":"2024-04-04T11:25:35.816650Z","shell.execute_reply":"2024-04-04T11:25:35.820488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split2path = {\n    \"train\": train_data_dir,\n    \"test\": test_data_dir,\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:35.822783Z","iopub.execute_input":"2024-04-04T11:25:35.823039Z","iopub.status.idle":"2024-04-04T11:25:35.832662Z","shell.execute_reply.started":"2024-04-04T11:25:35.823016Z","shell.execute_reply":"2024-04-04T11:25:35.831701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(data_path)\ndata.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:35.835380Z","iopub.execute_input":"2024-04-04T11:25:35.835834Z","iopub.status.idle":"2024-04-04T11:25:36.055345Z","shell.execute_reply.started":"2024-04-04T11:25:35.835800Z","shell.execute_reply":"2024-04-04T11:25:36.054362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_split(filename):\n    filename_ = filename.split(\"_\")\n    split = filename_[0]\n    return split\n\ndef extract_district(filename):\n    filename_ = filename.split(\" \")[0]\n    district = filename_.split(\"_\")[1]\n    return district\n\ndef beautify_dataset(data):\n    splits = []\n    districts = []\n    newpaths = []\n    transcripts = []\n    \n    for i in range(len(data)):\n        filename, transcript = data.iloc[i]\n        split = extract_split(filename)\n        district = extract_district(filename)\n        dir_path = split2path[split]\n        composed_path = f\"{dir_path}{filename}\"\n        \n        if os.path.exists(composed_path) == False:\n            print(f\"{composed_path} does not exist.\")\n            continue\n        \n        # replace any newline characters\n        transcript = transcript.replace(\"\\n\", \" \")\n        transcript = \" \".join(transcript.split())\n        \n        splits.append(split)\n        districts.append(district)\n        newpaths.append(composed_path)\n        transcripts.append(transcript)\n    \n    data['file_path'] = newpaths\n    data['district'] = districts\n    data['split'] = splits\n    data['transcripts'] = transcripts\n    \n#     data.drop(columns=['file_name'], inplace=True)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:36.056472Z","iopub.execute_input":"2024-04-04T11:25:36.056765Z","iopub.status.idle":"2024-04-04T11:25:36.066053Z","shell.execute_reply.started":"2024-04-04T11:25:36.056741Z","shell.execute_reply":"2024-04-04T11:25:36.065133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = beautify_dataset(data)\ndata.sample(20)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:36.067354Z","iopub.execute_input":"2024-04-04T11:25:36.067688Z","iopub.status.idle":"2024-04-04T11:25:58.167493Z","shell.execute_reply.started":"2024-04-04T11:25:36.067659Z","shell.execute_reply":"2024-04-04T11:25:58.166472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"<>\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.168811Z","iopub.execute_input":"2024-04-04T11:25:58.169128Z","iopub.status.idle":"2024-04-04T11:25:58.184756Z","shell.execute_reply.started":"2024-04-04T11:25:58.169074Z","shell.execute_reply":"2024-04-04T11:25:58.183549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.186192Z","iopub.execute_input":"2024-04-04T11:25:58.186541Z","iopub.status.idle":"2024-04-04T11:25:58.203518Z","shell.execute_reply.started":"2024-04-04T11:25:58.186511Z","shell.execute_reply":"2024-04-04T11:25:58.202643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"..\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.204688Z","iopub.execute_input":"2024-04-04T11:25:58.205017Z","iopub.status.idle":"2024-04-04T11:25:58.221436Z","shell.execute_reply.started":"2024-04-04T11:25:58.204987Z","shell.execute_reply":"2024-04-04T11:25:58.220599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTE:** Think of how you want use the existing models/your finetuned model to replace these examples.... For now let's just handle them.","metadata":{}},{"cell_type":"code","source":"# print(list(data[data['transcripts'] == ''].index))\ndata.drop(data[data['transcripts'] == ''].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '<>'].index))\ndata.drop(data[data['transcripts'] == \"<>\"].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '..'].index))\ndata.drop(data[data['transcripts'] == \"..\"].index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.225531Z","iopub.execute_input":"2024-04-04T11:25:58.225801Z","iopub.status.idle":"2024-04-04T11:25:58.252813Z","shell.execute_reply.started":"2024-04-04T11:25:58.225779Z","shell.execute_reply":"2024-04-04T11:25:58.251807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"] = data[\"transcripts\"].str.strip()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.253833Z","iopub.execute_input":"2024-04-04T11:25:58.254177Z","iopub.status.idle":"2024-04-04T11:25:58.263436Z","shell.execute_reply.started":"2024-04-04T11:25:58.254151Z","shell.execute_reply":"2024-04-04T11:25:58.262448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TASK = \"transcribe\"\nMODEL_NAME = \"openai/whisper-small\"","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.264473Z","iopub.execute_input":"2024-04-04T11:25:58.264762Z","iopub.status.idle":"2024-04-04T11:25:58.272867Z","shell.execute_reply.started":"2024-04-04T11:25:58.264737Z","shell.execute_reply":"2024-04-04T11:25:58.271974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\ntokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME, language='bn', task=TASK)\nprocessor = WhisperProcessor.from_pretrained(MODEL_NAME, language='bn', task=TASK)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:25:58.274100Z","iopub.execute_input":"2024-04-04T11:25:58.274492Z","iopub.status.idle":"2024-04-04T11:26:00.496490Z","shell.execute_reply.started":"2024-04-04T11:25:58.274462Z","shell.execute_reply":"2024-04-04T11:26:00.495653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = tokenizer.encode(\"\")\nids","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.497749Z","iopub.execute_input":"2024-04-04T11:26:00.498146Z","iopub.status.idle":"2024-04-04T11:26:00.505379Z","shell.execute_reply.started":"2024-04-04T11:26:00.498086Z","shell.execute_reply":"2024-04-04T11:26:00.504210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.decode(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.506492Z","iopub.execute_input":"2024-04-04T11:26:00.506761Z","iopub.status.idle":"2024-04-04T11:26:00.539006Z","shell.execute_reply.started":"2024-04-04T11:26:00.506737Z","shell.execute_reply":"2024-04-04T11:26:00.538138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n        \n        torch.cuda.empty_cache()\n\n        return batch","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.540182Z","iopub.execute_input":"2024-04-04T11:26:00.540461Z","iopub.status.idle":"2024-04-04T11:26:00.551198Z","shell.execute_reply.started":"2024-04-04T11:26:00.540420Z","shell.execute_reply":"2024-04-04T11:26:00.550285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.552258Z","iopub.execute_input":"2024-04-04T11:26:00.554263Z","iopub.status.idle":"2024-04-04T11:26:00.560424Z","shell.execute_reply.started":"2024-04-04T11:26:00.554225Z","shell.execute_reply":"2024-04-04T11:26:00.559627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(example):\n    audio_path = example[\"file_path\"]\n    \n    # load the audio using librosa or torch audio (as you wish)\n    audio, sr = librosa.load(audio_path, sr=16_000)\n    \n    example[\"input_features\"] = feature_extractor(audio, sampling_rate=sr).input_features[0]\n    \n    example[\"labels\"] = tokenizer(f\"{example['transcripts']}\", max_length=448, padding=True, truncation=True).input_ids\n    \n    return example\n\n\ndef filter_inputs(input_audio):\n    \"\"\"filter inputs with zero input length\"\"\"\n    return 0 < len(input_audio)\n\n\ndef filter_labels(input_labels):\n    \"\"\"filter empty label sequences\"\"\"\n    return 0 < len(input_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.561485Z","iopub.execute_input":"2024-04-04T11:26:00.561750Z","iopub.status.idle":"2024-04-04T11:26:00.572791Z","shell.execute_reply.started":"2024-04-04T11:26:00.561727Z","shell.execute_reply":"2024-04-04T11:26:00.571920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = data[data[\"split\"] == \"train\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.573861Z","iopub.execute_input":"2024-04-04T11:26:00.574934Z","iopub.status.idle":"2024-04-04T11:26:00.596479Z","shell.execute_reply.started":"2024-04-04T11:26:00.574905Z","shell.execute_reply":"2024-04-04T11:26:00.595511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    adjust test size accordingly.\n\"\"\"\ntrain_df, eval_df = train_test_split(train_df, test_size=0.01, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.597643Z","iopub.execute_input":"2024-04-04T11:26:00.598028Z","iopub.status.idle":"2024-04-04T11:26:00.606107Z","shell.execute_reply.started":"2024-04-04T11:26:00.597993Z","shell.execute_reply":"2024-04-04T11:26:00.605189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df), len(eval_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.607410Z","iopub.execute_input":"2024-04-04T11:26:00.607729Z","iopub.status.idle":"2024-04-04T11:26:00.615576Z","shell.execute_reply.started":"2024-04-04T11:26:00.607682Z","shell.execute_reply":"2024-04-04T11:26:00.614543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ben_reg_voice_ds = DatasetDict()\n\ntrain_split = DS.from_pandas(train_df)\neval_split = DS.from_pandas(eval_df)\n\nds_splits = DatasetDict({\n    'train': train_split,\n    'eval': eval_split\n})","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.616902Z","iopub.execute_input":"2024-04-04T11:26:00.617258Z","iopub.status.idle":"2024-04-04T11:26:00.684812Z","shell.execute_reply.started":"2024-04-04T11:26:00.617222Z","shell.execute_reply":"2024-04-04T11:26:00.683998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.remove_columns([\"split\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.686027Z","iopub.execute_input":"2024-04-04T11:26:00.686368Z","iopub.status.idle":"2024-04-04T11:26:00.694064Z","shell.execute_reply.started":"2024-04-04T11:26:00.686341Z","shell.execute_reply":"2024-04-04T11:26:00.693013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ds_splits)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.695174Z","iopub.execute_input":"2024-04-04T11:26:00.695450Z","iopub.status.idle":"2024-04-04T11:26:00.704512Z","shell.execute_reply.started":"2024-04-04T11:26:00.695427Z","shell.execute_reply":"2024-04-04T11:26:00.703628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.object = object","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.705892Z","iopub.execute_input":"2024-04-04T11:26:00.706353Z","iopub.status.idle":"2024-04-04T11:26:00.713796Z","shell.execute_reply.started":"2024-04-04T11:26:00.706296Z","shell.execute_reply":"2024-04-04T11:26:00.712914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.map(prepare_dataset, remove_columns=ds_splits.column_names[\"train\"],\n                          num_proc=2 # open for multithreadding\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:26:00.714949Z","iopub.execute_input":"2024-04-04T11:26:00.715236Z","iopub.status.idle":"2024-04-04T11:31:51.110088Z","shell.execute_reply.started":"2024-04-04T11:26:00.715212Z","shell.execute_reply":"2024-04-04T11:31:51.109115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ds_splits = ds_splits.filter(filter_inputs, input_columns=[\"input_features\"])\n# ds_splits = ds_splits.filter(filter_labels, input_columns=[\"labels\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:51.111685Z","iopub.execute_input":"2024-04-04T11:31:51.112548Z","iopub.status.idle":"2024-04-04T11:31:51.116787Z","shell.execute_reply.started":"2024-04-04T11:31:51.112510Z","shell.execute_reply":"2024-04-04T11:31:51.115903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ds_splits[\"train\"]), len(ds_splits[\"eval\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:51.122870Z","iopub.execute_input":"2024-04-04T11:31:51.123181Z","iopub.status.idle":"2024-04-04T11:31:51.133042Z","shell.execute_reply.started":"2024-04-04T11:31:51.123157Z","shell.execute_reply":"2024-04-04T11:31:51.132133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cer = CharErrorRate()\nwer = WordErrorRate()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:51.134069Z","iopub.execute_input":"2024-04-04T11:31:51.134339Z","iopub.status.idle":"2024-04-04T11:31:51.151741Z","shell.execute_reply.started":"2024-04-04T11:31:51.134318Z","shell.execute_reply":"2024-04-04T11:31:51.150794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    label_ids[label_ids == -100] = tokenizer.pad_token_id\n\n    pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = tokenizer.batch_decode(label_ids, skip_special_tokens=True)\n\n    wer_res = wer(pred_str, label_str)\n    cer_res = cer(pred_str, label_str)\n    \n    \"\"\"\n        uncomment the next 3 lines if you want to see how the examples look like during eval \n    \"\"\"\n    print(\"WER:\",wer_res,\"| CER:\", cer_res) # to show up during running logs\n    print(\"Pred:\",pred_str[0])\n    print(\"Label:\",label_str[0])\n    \n    return {\"wer\": wer_res, \"cer\": cer_res}","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:51.152874Z","iopub.execute_input":"2024-04-04T11:31:51.153189Z","iopub.status.idle":"2024-04-04T11:31:51.163946Z","shell.execute_reply.started":"2024-04-04T11:31:51.153165Z","shell.execute_reply":"2024-04-04T11:31:51.163128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, device_map=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:51.164957Z","iopub.execute_input":"2024-04-04T11:31:51.165290Z","iopub.status.idle":"2024-04-04T11:31:58.007375Z","shell.execute_reply.started":"2024-04-04T11:31:51.165266Z","shell.execute_reply":"2024-04-04T11:31:58.006468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_id = \"whisper-reg-ben\"","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:58.008507Z","iopub.execute_input":"2024-04-04T11:31:58.008793Z","iopub.status.idle":"2024-04-04T11:31:58.013095Z","shell.execute_reply.started":"2024-04-04T11:31:58.008768Z","shell.execute_reply":"2024-04-04T11:31:58.012060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = Seq2SeqTrainingArguments(\n    output_dir=model_id,\n    per_device_train_batch_size=9,\n    per_device_eval_batch_size=8,\n    gradient_accumulation_steps=1,\n    gradient_checkpointing=True,\n    fp16=True,\n    learning_rate=3e-4,\n    weight_decay=1e-2,\n    warmup_steps=100,\n    num_train_epochs=1,\n    evaluation_strategy=\"steps\", # or \"epochs\"\n    predict_with_generate=True,\n#     generation_max_length=448,\n    save_steps=1000,\n    eval_steps=1000,\n    logging_steps=1000,\n    save_total_limit=1,\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    push_to_hub=False,\n    report_to=\"none\",\n    remove_unused_columns=False,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:58.014449Z","iopub.execute_input":"2024-04-04T11:31:58.015201Z","iopub.status.idle":"2024-04-04T11:31:58.026809Z","shell.execute_reply.started":"2024-04-04T11:31:58.015168Z","shell.execute_reply":"2024-04-04T11:31:58.026073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.generation_config.language = \"bn\"\nmodel.generation_config.task = \"transcribe\"\n\nmodel.generation_config.forced_decoder_ids = None\nmodel.config.suppress_tokens = [] # added later","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:58.027840Z","iopub.execute_input":"2024-04-04T11:31:58.028118Z","iopub.status.idle":"2024-04-04T11:31:58.036894Z","shell.execute_reply.started":"2024-04-04T11:31:58.028095Z","shell.execute_reply":"2024-04-04T11:31:58.035905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=ds_splits[\"train\"],\n    eval_dataset=ds_splits[\"eval\"],\n    data_collator=data_collator,\n    tokenizer=processor.feature_extractor,\n    compute_metrics=compute_metrics,\n#     callbacks=[EarlyStoppingCallback(2, 1.0)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:58.038236Z","iopub.execute_input":"2024-04-04T11:31:58.038606Z","iopub.status.idle":"2024-04-04T11:31:58.059746Z","shell.execute_reply.started":"2024-04-04T11:31:58.038574Z","shell.execute_reply":"2024-04-04T11:31:58.058688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()\n\n# to use the high-level pipeline, ensure both the processor outputs and model outputs exist in the same dir\ntrainer.save_model(training_args.output_dir)\nprocessor.save_pretrained(training_args.output_dir)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T11:31:58.060806Z","iopub.execute_input":"2024-04-04T11:31:58.061349Z","iopub.status.idle":"2024-04-04T13:34:23.629640Z","shell.execute_reply.started":"2024-04-04T11:31:58.061318Z","shell.execute_reply":"2024-04-04T13:34:23.628701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"out_logs = pd.DataFrame(trainer.state.log_history)\nout_logs.to_csv(\"logs.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:23.630910Z","iopub.execute_input":"2024-04-04T13:34:23.631231Z","iopub.status.idle":"2024-04-04T13:34:23.697610Z","shell.execute_reply.started":"2024-04-04T13:34:23.631205Z","shell.execute_reply":"2024-04-04T13:34:23.696877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel ds_splits\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:23.698724Z","iopub.execute_input":"2024-04-04T13:34:23.699097Z","iopub.status.idle":"2024-04-04T13:34:24.028138Z","shell.execute_reply.started":"2024-04-04T13:34:23.699043Z","shell.execute_reply":"2024-04-04T13:34:24.027150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:24.029300Z","iopub.execute_input":"2024-04-04T13:34:24.029579Z","iopub.status.idle":"2024-04-04T13:34:24.081011Z","shell.execute_reply.started":"2024-04-04T13:34:24.029555Z","shell.execute_reply":"2024-04-04T13:34:24.080024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe = pipeline(\n    \"automatic-speech-recognition\",\n    model=model_id,\n    chunk_length_s=30,\n    device=0,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:24.082609Z","iopub.execute_input":"2024-04-04T13:34:24.082952Z","iopub.status.idle":"2024-04-04T13:34:25.335151Z","shell.execute_reply.started":"2024-04-04T13:34:24.082902Z","shell.execute_reply":"2024-04-04T13:34:25.333952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pretty_sort(filename):\n    name, number_str = filename.split(\" (\")\n    number = int(number_str.split(\")\")[0])\n    return name, number","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:25.383880Z","iopub.execute_input":"2024-04-04T13:34:25.384197Z","iopub.status.idle":"2024-04-04T13:34:25.391790Z","shell.execute_reply.started":"2024-04-04T13:34:25.384173Z","shell.execute_reply":"2024-04-04T13:34:25.390874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:25.393135Z","iopub.execute_input":"2024-04-04T13:34:25.393479Z","iopub.status.idle":"2024-04-04T13:34:25.404471Z","shell.execute_reply.started":"2024-04-04T13:34:25.393449Z","shell.execute_reply":"2024-04-04T13:34:25.403604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for root, dirs, files in os.walk(\"/kaggle/input/ben10/ben10/16_kHz_valid_audio\"):\n    files = sorted(files, key=pretty_sort)\n    \n#     print(files.index(\"valid_sandwip (1).wav\"))\n#     print(files.index(\"valid_sandwip (132).wav\"))\n    \n#     put swandip first\n    shift = files[1070 : 1202]\n    \n    files = shift + files[:1070] + files[1202:]\n    ids = files.copy()\n    \n    for file in files:\n        composed_path = f\"{test_data_dir}{file}\"\n        audio, sr = librosa.load(composed_path, sr=16_000)\n        text = pipe(audio)[\"text\"]\n        preds.append(text)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:34:25.405588Z","iopub.execute_input":"2024-04-04T13:34:25.406252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[\"id\"] = ids\nsub_df[\"sentence\"] = preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}