{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":6707460,"sourceType":"datasetVersion","datasetId":3865741}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install -q bitsandbytes #accelerate bitsandbytes==0.37\n# !pip install -q git+https://github.com/huggingface/peft.git@main\n# !pip install -q git+https://github.com/huggingface/accelerate.git@main\n# !pip install -q noisereduce","metadata":{"execution":{"iopub.status.busy":"2024-04-07T12:05:14.345870Z","iopub.execute_input":"2024-04-07T12:05:14.346180Z","iopub.status.idle":"2024-04-07T12:05:14.350736Z","shell.execute_reply.started":"2024-04-07T12:05:14.346150Z","shell.execute_reply":"2024-04-07T12:05:14.349651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install jiwer\n!pip install bnlp-toolkit\n!pip install banglanum2words\n!pip install git+https://github.com/csebuetnlp/normalizer","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:50:30.533433Z","iopub.execute_input":"2024-04-19T06:50:30.533795Z","iopub.status.idle":"2024-04-19T06:51:43.225250Z","shell.execute_reply.started":"2024-04-19T06:50:30.533769Z","shell.execute_reply":"2024-04-19T06:51:43.224128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport librosa\nfrom tqdm import tqdm\nimport numpy as np\nimport random\nfrom pydub import AudioSegment\nimport librosa\nimport matplotlib.pyplot as plt\nimport os\nfrom multiprocessing import Pool\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:51:43.226943Z","iopub.execute_input":"2024-04-19T06:51:43.227233Z","iopub.status.idle":"2024-04-19T06:51:44.114957Z","shell.execute_reply.started":"2024-04-19T06:51:43.227207Z","shell.execute_reply":"2024-04-19T06:51:44.114170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:51:48.080603Z","iopub.execute_input":"2024-04-19T06:51:48.081734Z","iopub.status.idle":"2024-04-19T06:51:49.129172Z","shell.execute_reply.started":"2024-04-19T06:51:48.081704Z","shell.execute_reply":"2024-04-19T06:51:49.128185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchaudio\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nfrom datasets import DatasetDict\nfrom datasets import Dataset as DS\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    BitsAndBytesConfig,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer, \n    TrainerCallback, \n    TrainingArguments, \n    TrainerState, \n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline\n)\n\n\nfrom torchmetrics.text import WordErrorRate, CharErrorRate\n","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:51:49.131195Z","iopub.execute_input":"2024-04-19T06:51:49.131930Z","iopub.status.idle":"2024-04-19T06:52:09.104719Z","shell.execute_reply.started":"2024-04-19T06:51:49.131896Z","shell.execute_reply":"2024-04-19T06:52:09.103870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/ben10/ben10'\ntrain_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ntest_data_dir = f\"{BASE_DIR}/16_kHz_valid_audio/\"\ndata_path = f\"{BASE_DIR}/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:09.105851Z","iopub.execute_input":"2024-04-19T06:52:09.106394Z","iopub.status.idle":"2024-04-19T06:52:09.111348Z","shell.execute_reply.started":"2024-04-19T06:52:09.106371Z","shell.execute_reply":"2024-04-19T06:52:09.110287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split2path = {\n    \"train\": train_data_dir,\n    \"test\": test_data_dir,\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:09.113894Z","iopub.execute_input":"2024-04-19T06:52:09.114563Z","iopub.status.idle":"2024-04-19T06:52:09.135757Z","shell.execute_reply.started":"2024-04-19T06:52:09.114503Z","shell.execute_reply":"2024-04-19T06:52:09.134950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(data_path)\ndata.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:09.136847Z","iopub.execute_input":"2024-04-19T06:52:09.137216Z","iopub.status.idle":"2024-04-19T06:52:09.378775Z","shell.execute_reply.started":"2024-04-19T06:52:09.137192Z","shell.execute_reply":"2024-04-19T06:52:09.377850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_split(filename):\n    filename_ = filename.split(\"_\")\n    split = filename_[0]\n    return split\n\ndef extract_district(filename):\n    filename_ = filename.split(\" \")[0]\n    district = filename_.split(\"_\")[1]\n    return district\n\ndef beautify_dataset(data):\n    splits = []\n    districts = []\n    newpaths = []\n    transcripts = []\n    \n    for i in range(len(data)):\n        filename, transcript = data.iloc[i]\n        split = extract_split(filename)\n        district = extract_district(filename)\n        dir_path = split2path[split]\n        composed_path = f\"{dir_path}{filename}\"\n        \n        if os.path.exists(composed_path) == False:\n            print(f\"{composed_path} does not exist.\")\n            continue\n        \n        # replace any newline characters\n        transcript = transcript.replace(\"\\n\", \" \")\n        transcript = \" \".join(transcript.split())\n        \n        splits.append(split)\n        districts.append(district)\n        newpaths.append(composed_path)\n        transcripts.append(transcript)\n    \n    data['file_path'] = newpaths\n    data['district'] = districts\n    data['split'] = splits\n    data['transcripts'] = transcripts\n    \n#     data.drop(columns=['file_name'], inplace=True)\n    \n    return data\n","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:09.380220Z","iopub.execute_input":"2024-04-19T06:52:09.380516Z","iopub.status.idle":"2024-04-19T06:52:09.389862Z","shell.execute_reply.started":"2024-04-19T06:52:09.380492Z","shell.execute_reply":"2024-04-19T06:52:09.388912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = beautify_dataset(data)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:09.390845Z","iopub.execute_input":"2024-04-19T06:52:09.391091Z","iopub.status.idle":"2024-04-19T06:52:58.051738Z","shell.execute_reply.started":"2024-04-19T06:52:09.391071Z","shell.execute_reply":"2024-04-19T06:52:58.050883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.sample(20)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.053091Z","iopub.execute_input":"2024-04-19T06:52:58.053357Z","iopub.status.idle":"2024-04-19T06:52:58.069766Z","shell.execute_reply.started":"2024-04-19T06:52:58.053334Z","shell.execute_reply":"2024-04-19T06:52:58.068777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.groupby(\"district\").sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.073294Z","iopub.execute_input":"2024-04-19T06:52:58.073758Z","iopub.status.idle":"2024-04-19T06:52:58.391460Z","shell.execute_reply.started":"2024-04-19T06:52:58.073717Z","shell.execute_reply":"2024-04-19T06:52:58.390518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"<>\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.392946Z","iopub.execute_input":"2024-04-19T06:52:58.393294Z","iopub.status.idle":"2024-04-19T06:52:58.410134Z","shell.execute_reply.started":"2024-04-19T06:52:58.393263Z","shell.execute_reply":"2024-04-19T06:52:58.409102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.411600Z","iopub.execute_input":"2024-04-19T06:52:58.412279Z","iopub.status.idle":"2024-04-19T06:52:58.428549Z","shell.execute_reply.started":"2024-04-19T06:52:58.412254Z","shell.execute_reply":"2024-04-19T06:52:58.427499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \".\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.429846Z","iopub.execute_input":"2024-04-19T06:52:58.430183Z","iopub.status.idle":"2024-04-19T06:52:58.444267Z","shell.execute_reply.started":"2024-04-19T06:52:58.430156Z","shell.execute_reply":"2024-04-19T06:52:58.443264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data[\"transcripts\"] == \"..\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.445430Z","iopub.execute_input":"2024-04-19T06:52:58.445803Z","iopub.status.idle":"2024-04-19T06:52:58.462364Z","shell.execute_reply.started":"2024-04-19T06:52:58.445772Z","shell.execute_reply":"2024-04-19T06:52:58.461495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(list(data[data['transcripts'] == ''].index))\ndata.drop(data[data['transcripts'] == ''].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '<>'].index))\ndata.drop(data[data['transcripts'] == \"<>\"].index, inplace=True)\n      \n# print(list(data[data['transcripts'] == '..'].index))\ndata.drop(data[data['transcripts'] == \"..\"].index, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.463541Z","iopub.execute_input":"2024-04-19T06:52:58.463863Z","iopub.status.idle":"2024-04-19T06:52:58.491347Z","shell.execute_reply.started":"2024-04-19T06:52:58.463833Z","shell.execute_reply":"2024-04-19T06:52:58.490565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"].str.len().idxmin(), data[\"transcripts\"].str.len().idxmax(), data[\"transcripts\"].str.len().min(), data[\"transcripts\"].str.len().max()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.492630Z","iopub.execute_input":"2024-04-19T06:52:58.493014Z","iopub.status.idle":"2024-04-19T06:52:58.534570Z","shell.execute_reply.started":"2024-04-19T06:52:58.492978Z","shell.execute_reply":"2024-04-19T06:52:58.533449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"].info()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.535802Z","iopub.execute_input":"2024-04-19T06:52:58.536142Z","iopub.status.idle":"2024-04-19T06:52:58.552618Z","shell.execute_reply.started":"2024-04-19T06:52:58.536105Z","shell.execute_reply":"2024-04-19T06:52:58.551724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.553733Z","iopub.execute_input":"2024-04-19T06:52:58.554026Z","iopub.status.idle":"2024-04-19T06:52:58.568636Z","shell.execute_reply.started":"2024-04-19T06:52:58.554003Z","shell.execute_reply":"2024-04-19T06:52:58.567662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.569905Z","iopub.execute_input":"2024-04-19T06:52:58.570248Z","iopub.status.idle":"2024-04-19T06:52:58.580432Z","shell.execute_reply.started":"2024-04-19T06:52:58.570216Z","shell.execute_reply":"2024-04-19T06:52:58.579573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"transcripts\"] = data[\"transcripts\"].str.strip()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.581459Z","iopub.execute_input":"2024-04-19T06:52:58.581756Z","iopub.status.idle":"2024-04-19T06:52:58.593507Z","shell.execute_reply.started":"2024-04-19T06:52:58.581734Z","shell.execute_reply":"2024-04-19T06:52:58.592645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport unicodedata\nfrom normalizer import normalize\nfrom banglanum2words import num_convert\n\ndef unicode_to_ascii(s):\n    return unicodedata.normalize('NFKC', s)\n\ndef is_english_numeral(numeral):\n    english_numeral_regex = re.compile(\"[0-9]\")\n    return english_numeral_regex.search(numeral) is not None\n\ndef convert_numbers_to_words(text):\n    numerals = re.findall(r'\\d+', text)\n    bangla_numerals = []\n    for ix,numeral in enumerate(numerals):\n        if not is_english_numeral(numeral):\n            bangla_numerals.append(numeral)\n    for numeral in bangla_numerals:\n        text = re.sub(numeral, num_convert.number_to_bangla_words(str(numeral)) ,text)\n    return text\n\n\ndef remove_chars(text,remove_english=True):\n    #chars_to_ignore = '[{(-:;\\'\"¿!\\?\\|)}]'\n    chars_to_ignore = '[{(।,/:;.\\'\"¿!*\\?\\-|)}]'\n    text = re.sub(chars_to_ignore, \" \", text)\n    if remove_english:\n        text = re.sub(r\"[a-zA-Z]+\", \"\", text)\n        text = re.sub(r\"[0-9]+\", \"\", text)\n    text = re.sub(\"\\t\",\"\", text)\n    return text\n\ndef remove_extra_whitespace(sentence):\n    words = sentence.split()\n    return \" \".join(words)\n\ndef preprocess_sentence(s,remove_english=True):\n    s = s.lower().strip()\n    s = remove_chars(s, remove_english)\n    #s = replace_numerals_with_words(s)\n    #s = convert_numbers_to_words(s)\n    s = normalize(s)\n    s = unicode_to_ascii(s)\n    s = remove_extra_whitespace(s)\n    return s","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.594734Z","iopub.execute_input":"2024-04-19T06:52:58.595080Z","iopub.status.idle":"2024-04-19T06:52:58.780016Z","shell.execute_reply.started":"2024-04-19T06:52:58.595051Z","shell.execute_reply":"2024-04-19T06:52:58.778992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.copy()\ndata[\"transcripts\"] = data[\"transcripts\"].apply(preprocess_sentence)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:52:58.781319Z","iopub.execute_input":"2024-04-19T06:52:58.781632Z","iopub.status.idle":"2024-04-19T06:53:07.470366Z","shell.execute_reply.started":"2024-04-19T06:52:58.781607Z","shell.execute_reply":"2024-04-19T06:53:07.469556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TASK = \"transcribe\"\nMODEL_NAME = \"openai/whisper-medium\"\nMODEL_NAME = \"bangla-speech-processing/BanglaASR\"\nMODEL_NAME = \"/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium\"","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:53:07.471466Z","iopub.execute_input":"2024-04-19T06:53:07.471774Z","iopub.status.idle":"2024-04-19T06:53:07.476668Z","shell.execute_reply.started":"2024-04-19T06:53:07.471748Z","shell.execute_reply":"2024-04-19T06:53:07.475555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\n# tokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME, language='bn', task=TASKMOOL_NAME, language='bn', task=TASK)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:24.616106Z","iopub.execute_input":"2024-04-08T17:32:24.616498Z","iopub.status.idle":"2024-04-08T17:32:24.627728Z","shell.execute_reply.started":"2024-04-08T17:32:24.616429Z","shell.execute_reply":"2024-04-08T17:32:24.626908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\ntokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME)\nprocessor = WhisperProcessor.from_pretrained(MODEL_NAME)\nmodel = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, device_map=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:53:07.480046Z","iopub.execute_input":"2024-04-19T06:53:07.480316Z","iopub.status.idle":"2024-04-19T06:53:59.378375Z","shell.execute_reply.started":"2024-04-19T06:53:07.480293Z","shell.execute_reply":"2024-04-19T06:53:59.377574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = tokenizer.encode(\"\")\nids","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:53:59.379924Z","iopub.execute_input":"2024-04-19T06:53:59.380200Z","iopub.status.idle":"2024-04-19T06:53:59.386380Z","shell.execute_reply.started":"2024-04-19T06:53:59.380176Z","shell.execute_reply":"2024-04-19T06:53:59.385435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.decode(ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:53:59.387517Z","iopub.execute_input":"2024-04-19T06:53:59.387802Z","iopub.status.idle":"2024-04-19T06:53:59.397645Z","shell.execute_reply.started":"2024-04-19T06:53:59.387779Z","shell.execute_reply":"2024-04-19T06:53:59.396753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n        \n        torch.cuda.empty_cache()\n        \n#         print(batch)\n# #         print(batch.shape)\n#         for x in batch[\"input_features\"]: \n#             print(x.shape)\n#         print(batch[\"labels\"].shape)\n\n        return batch","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:54:25.636831Z","iopub.execute_input":"2024-04-19T06:54:25.637614Z","iopub.status.idle":"2024-04-19T06:54:25.647268Z","shell.execute_reply.started":"2024-04-19T06:54:25.637586Z","shell.execute_reply":"2024-04-19T06:54:25.646100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:54:29.234739Z","iopub.execute_input":"2024-04-19T06:54:29.235460Z","iopub.status.idle":"2024-04-19T06:54:29.239597Z","shell.execute_reply.started":"2024-04-19T06:54:29.235433Z","shell.execute_reply":"2024-04-19T06:54:29.238502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(example):\n    audio_path = example[\"file_path\"]\n    \n    audio, sr = librosa.load(audio_path, sr=16_000)\n    \n    example[\"input_features\"] = feature_extractor(audio, sampling_rate=sr).input_features[0]\n    \n    example[\"labels\"] = tokenizer(f\"{example['transcripts']}\", max_length=448, padding=True, truncation=True).input_ids\n    \n    return example\n\n\ndef filter_inputs(input_audio):\n    \"\"\"Filter inputs with zero input length\"\"\"\n    return 0 < len(input_audio)\n\n\ndef filter_labels(input_labels):\n    \"\"\"Filter empty label sequences\"\"\"\n    return 0 < len(input_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:54:30.118557Z","iopub.execute_input":"2024-04-19T06:54:30.119209Z","iopub.status.idle":"2024-04-19T06:54:30.126096Z","shell.execute_reply.started":"2024-04-19T06:54:30.119179Z","shell.execute_reply":"2024-04-19T06:54:30.125055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = data[data[\"split\"] == \"train\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.668338Z","iopub.execute_input":"2024-04-08T17:32:32.668587Z","iopub.status.idle":"2024-04-08T17:32:32.686482Z","shell.execute_reply.started":"2024-04-08T17:32:32.668565Z","shell.execute_reply":"2024-04-08T17:32:32.685668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, eval_df = train_test_split(train_df, test_size=0.01, shuffle=True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.687623Z","iopub.execute_input":"2024-04-08T17:32:32.687970Z","iopub.status.idle":"2024-04-08T17:32:32.696226Z","shell.execute_reply.started":"2024-04-08T17:32:32.687919Z","shell.execute_reply":"2024-04-08T17:32:32.695397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df), len(eval_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.697302Z","iopub.execute_input":"2024-04-08T17:32:32.697664Z","iopub.status.idle":"2024-04-08T17:32:32.704109Z","shell.execute_reply.started":"2024-04-08T17:32:32.697577Z","shell.execute_reply":"2024-04-08T17:32:32.703283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ben_reg_voice_ds = DatasetDict()\n\ntrain_split = DS.from_pandas(train_df)\neval_split = DS.from_pandas(eval_df)\n\nds_splits = DatasetDict({\n    'train': train_split,\n    'eval': eval_split\n})","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.705133Z","iopub.execute_input":"2024-04-08T17:32:32.705445Z","iopub.status.idle":"2024-04-08T17:32:32.764215Z","shell.execute_reply.started":"2024-04-08T17:32:32.705414Z","shell.execute_reply":"2024-04-08T17:32:32.763233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.remove_columns([\"split\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.765315Z","iopub.execute_input":"2024-04-08T17:32:32.765576Z","iopub.status.idle":"2024-04-08T17:32:32.772840Z","shell.execute_reply.started":"2024-04-08T17:32:32.765553Z","shell.execute_reply":"2024-04-08T17:32:32.771755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ds_splits)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.773920Z","iopub.execute_input":"2024-04-08T17:32:32.774194Z","iopub.status.idle":"2024-04-08T17:32:32.784945Z","shell.execute_reply.started":"2024-04-08T17:32:32.774155Z","shell.execute_reply":"2024-04-08T17:32:32.783798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.object = object","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.785996Z","iopub.execute_input":"2024-04-08T17:32:32.786316Z","iopub.status.idle":"2024-04-08T17:32:32.793912Z","shell.execute_reply.started":"2024-04-08T17:32:32.786290Z","shell.execute_reply":"2024-04-08T17:32:32.793012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_splits = ds_splits.map(prepare_dataset, remove_columns=ds_splits.column_names[\"train\"], num_proc=2)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:32:32.794978Z","iopub.execute_input":"2024-04-08T17:32:32.795576Z","iopub.status.idle":"2024-04-08T17:38:31.606274Z","shell.execute_reply.started":"2024-04-08T17:32:32.795543Z","shell.execute_reply":"2024-04-08T17:38:31.605201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ds_splits[\"train\"]), len(ds_splits[\"eval\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:34.800974Z","iopub.execute_input":"2024-04-08T17:53:34.801808Z","iopub.status.idle":"2024-04-08T17:53:34.808906Z","shell.execute_reply.started":"2024-04-08T17:53:34.801766Z","shell.execute_reply":"2024-04-08T17:53:34.807935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cer = CharErrorRate()\nwer = WordErrorRate()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:35.056534Z","iopub.execute_input":"2024-04-08T17:53:35.056815Z","iopub.status.idle":"2024-04-08T17:53:35.065389Z","shell.execute_reply.started":"2024-04-08T17:53:35.056793Z","shell.execute_reply":"2024-04-08T17:53:35.064495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"domain_weights = {\n        'Barishal': 0.125,\n        'Chittagong': 0.083,\n        'Habiganj': 0.125,\n        'Kishoreganj': 0.083,\n        'Narail': 0.083, \n        'Narsingdi': 0.083,\n        'Rangpur': 0.083,\n        'Sylhet': 0.125,\n        'Sandwip': 0.125,\n        'Tangail': 0.083,\n    }\n\ndef compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    label_ids[label_ids == -100] = tokenizer.pad_token_id\n\n    pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = tokenizer.batch_decode(label_ids, skip_special_tokens=True)\n\n    wer_res = wer(pred_str, label_str)\n    cer_res = cer(pred_str, label_str)\n    \n    \"\"\"\n        uncomment the next 3 lines if you want to see how the examples look like during eval \n    \"\"\"\n#     print(\"WER:\",wer_res,\"| CER:\", cer_res)\n#     print(\"Pred:\",pred_str[0])\n#     print(\"Label:\",label_str[0])\n    \n    return {\"wer\": wer_res}","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:35.258206Z","iopub.execute_input":"2024-04-08T17:53:35.258493Z","iopub.status.idle":"2024-04-08T17:53:35.265613Z","shell.execute_reply.started":"2024-04-08T17:53:35.258469Z","shell.execute_reply":"2024-04-08T17:53:35.264656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# quant_config = BitsAndBytesConfig(\n#     load_in_4bit=True,\n#     bnb_4bit_quant_type=\"nf4\",\n#     bnb_4bit_use_double_quant=True,\n#     bnb_4bit_compute_dtype=torch.bfloat16,\n# )\n\n#model = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, quantization_config=quant_config, device_map=\"auto\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:36.002174Z","iopub.execute_input":"2024-04-08T17:53:36.002549Z","iopub.status.idle":"2024-04-08T17:53:36.006805Z","shell.execute_reply.started":"2024-04-08T17:53:36.002516Z","shell.execute_reply":"2024-04-08T17:53:36.005808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model = prepare_model_for_kbit_training(model)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:36.542399Z","iopub.execute_input":"2024-04-08T17:53:36.543150Z","iopub.status.idle":"2024-04-08T17:53:36.547109Z","shell.execute_reply.started":"2024-04-08T17:53:36.543115Z","shell.execute_reply":"2024-04-08T17:53:36.546052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"\"\"\n#     whisper uses CNNs in the encoder\n#     LoRA freezes all layers, but let's make the first layer (receiving layer) trainable.\n# \"\"\"\n# def make_inputs_require_grad(module, input, output):\n#     output.requires_grad_(True)\n\n# model.model.encoder.conv1.register_forward_hook(make_inputs_require_grad)\n# # model.model.encoder.conv2.register_forward_hook(make_inputs_require_grad)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:36.998492Z","iopub.execute_input":"2024-04-08T17:53:36.999223Z","iopub.status.idle":"2024-04-08T17:53:37.003142Z","shell.execute_reply.started":"2024-04-08T17:53:36.999190Z","shell.execute_reply":"2024-04-08T17:53:37.002186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_id = \"BanglaASR-reg\"","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:53:38.078277Z","iopub.execute_input":"2024-04-08T17:53:38.079256Z","iopub.status.idle":"2024-04-08T17:53:38.083139Z","shell.execute_reply.started":"2024-04-08T17:53:38.079219Z","shell.execute_reply":"2024-04-08T17:53:38.082199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 4\ntrain_batch_size = 32\neval_batch_size = 32\ntrain_steps_per_epoch = round(len(ds_splits[\"train\"])/train_batch_size)\neval_steps_per_epoch = round(len(ds_splits[\"eval\"])/eval_batch_size)\n# save_steps = round(train_steps_per_epoch*0.5)\neval_steps = 1000\nsave_steps  = eval_steps*3\n\nprint(f\"Total training Steps = {train_steps_per_epoch*num_epochs}\")\nprint(f\"Save frequency in steps = {save_steps}\")\nprint(f\"Evaluation frequency in steps = {eval_steps}\")\nprint(f\"Train steps per epoch = {train_steps_per_epoch}\")\nprint(f\"Validation steps per epoch = {eval_steps_per_epoch}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:54:57.599773Z","iopub.execute_input":"2024-04-08T17:54:57.600493Z","iopub.status.idle":"2024-04-08T17:54:57.607153Z","shell.execute_reply.started":"2024-04-08T17:54:57.600463Z","shell.execute_reply":"2024-04-08T17:54:57.605948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = Seq2SeqTrainingArguments(\n    output_dir=model_id,\n    per_device_train_batch_size=train_batch_size,\n    per_device_eval_batch_size=eval_batch_size,\n    gradient_accumulation_steps=4,\n    gradient_checkpointing=True,\n    fp16=True,\n    learning_rate=3e-4,\n    weight_decay=1e-2,\n    warmup_steps=50,\n    num_train_epochs=num_epochs,\n    evaluation_strategy=\"steps\", # or \"epochs\"\n    save_steps=save_steps,\n    eval_steps=eval_steps,\n    logging_steps=train_steps_per_epoch,\n    save_total_limit=1,\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    lr_scheduler_type=\"cosine_with_restarts\",\n    predict_with_generate=True,\n    #generation_max_length=448,\n    push_to_hub=False,\n    report_to=\"none\",\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:55:16.402281Z","iopub.execute_input":"2024-04-08T17:55:16.403116Z","iopub.status.idle":"2024-04-08T17:55:16.409781Z","shell.execute_reply.started":"2024-04-08T17:55:16.403083Z","shell.execute_reply":"2024-04-08T17:55:16.408757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.generation_config.language = \"bn\"\nmodel.generation_config.task = \"transcribe\"\n\nmodel.generation_config.forced_decoder_ids = None\nmodel.config.suppress_tokens = []","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:55:19.999762Z","iopub.execute_input":"2024-04-08T17:55:20.000128Z","iopub.status.idle":"2024-04-08T17:55:20.004961Z","shell.execute_reply.started":"2024-04-08T17:55:20.000097Z","shell.execute_reply":"2024-04-08T17:55:20.004011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=ds_splits[\"train\"],\n    eval_dataset=ds_splits[\"eval\"],\n    data_collator=data_collator,\n    tokenizer=processor.feature_extractor,\n    compute_metrics=compute_metrics,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:55:21.459404Z","iopub.execute_input":"2024-04-08T17:55:21.460196Z","iopub.status.idle":"2024-04-08T17:55:21.489526Z","shell.execute_reply.started":"2024-04-08T17:55:21.460166Z","shell.execute_reply":"2024-04-08T17:55:21.488659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()\n# to use the high-level pipeline, ensure both the processor outputs and model outputs exist in the same dir\ntrainer.save_model(training_args.output_dir)\nprocessor.save_pretrained(training_args.output_dir)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T17:55:55.328355Z","iopub.execute_input":"2024-04-08T17:55:55.328877Z","iopub.status.idle":"2024-04-08T20:02:44.705599Z","shell.execute_reply.started":"2024-04-08T17:55:55.328842Z","shell.execute_reply":"2024-04-08T20:02:44.704587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"out_logs = pd.DataFrame(trainer.state.log_history)\nout_logs.to_csv(\"logs.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:02:44.707119Z","iopub.execute_input":"2024-04-08T20:02:44.707415Z","iopub.status.idle":"2024-04-08T20:02:44.780055Z","shell.execute_reply.started":"2024-04-08T20:02:44.707389Z","shell.execute_reply":"2024-04-08T20:02:44.779259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel ds_splits\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:02:44.781106Z","iopub.execute_input":"2024-04-08T20:02:44.781455Z","iopub.status.idle":"2024-04-08T20:02:45.099238Z","shell.execute_reply.started":"2024-04-08T20:02:44.781419Z","shell.execute_reply":"2024-04-08T20:02:45.098317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:02:45.101806Z","iopub.execute_input":"2024-04-08T20:02:45.102448Z","iopub.status.idle":"2024-04-08T20:02:45.171369Z","shell.execute_reply.started":"2024-04-08T20:02:45.102412Z","shell.execute_reply":"2024-04-08T20:02:45.170410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pipe = pipeline(\n#     \"automatic-speech-recognition\",\n#     model=model,\n#     tokenizer=processor.tokenizer,\n#     feature_extractor=processor.feature_extractor, \n#     chunk_length_s=15, # chunk 15 secs\n#     torch_dtype=torch.float16\n# )\n\npipe = pipeline(\n    \"automatic-speech-recognition\",\n    model=model_id,\n    chunk_length_s=30,\n    device=0,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:02:45.193250Z","iopub.execute_input":"2024-04-08T20:02:45.193516Z","iopub.status.idle":"2024-04-08T20:02:46.300405Z","shell.execute_reply.started":"2024-04-08T20:02:45.193486Z","shell.execute_reply":"2024-04-08T20:02:46.299412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport time\nimport math\nimport torch\nimport matplotlib.pyplot as plt\nfrom typing import Dict,Tuple, Optional, Type\n\n\nclass ProgressBar:\n    \"\"\"A progress bar for tracking the progress of iterative tasks.\n\n    Displays a progress bar with percentage completion, elapsed time, estimated time remaining,\n    and optional verbose text.\n\n    **Example usage:**\n\n    .. code-block:: python\n\n        no_iter = 100\n        pb = ProgressBar(total=no_iter, prefix=\"Processing\", bar_length=50)\n        for i in range(no_iter):\n            pb.update(i + 1, message=f\"Processing item {i + 1}\")\n\n    \"\"\"\n\n    def __init__(self, total: int, bar_length: int = 30, fill: str = \"=\", prefix: str = \"\") -> None:\n        \"\"\"Initialize the instance of ProgressBar class.\n\n        Args:\n            total (int): Total number of iterations. Defaults to None.\n            bar_length (int, optional): Length of the progress bar. Defaults to ``30``.\n            fill (str, optional): Character used to fill the progress bar. Defaults to ``'='``.\n            prefix (str, optional): A text to display before the progress bar. Defaults to empty string ``\"\"``.\n        \"\"\"\n\n        self.total = total\n        self.length = bar_length\n        self.fill = fill\n        self.prefix = prefix + \": \" if prefix else \"\"\n        self.start_time = time.time()\n        self._prev_len = 0\n\n    def update(self, iteration: int, message: str = \"\") -> None:\n        \"\"\"\n        Update the progress bar with current iteration and optional message text.\n\n        Args:\n            iteration (int): Current iteration number.\n            message (str, optional): Optional text to display. Defaults to empty string.\n\n        \"\"\"\n\n        # Calculate progress and format bar\n        percent = (\"{0:.1f}\").format(100 * (iteration / float(self.total)))\n        filled_length = int(self.length * iteration // self.total)\n        bar = self.fill * filled_length + \"-\" * (self.length - filled_length)\n\n        # Calculate elapsed and estimated time\n        elapsed_time = time.time() - self.start_time\n        ET = self.format_time(elapsed_time)\n        estimated_time_remaining = (elapsed_time / iteration) * (self.total - iteration)\n        ETR = self.format_time(estimated_time_remaining)\n\n        # Print progress bar with appropriate text\n        if iteration == self.total:\n            step_time = self.format_time(elapsed_time / self.total)\n            info = f\"\\r{self.prefix}{iteration}/{self.total} [{bar}] {percent}% | ET: {ET} | {step_time}\\\\step | \"\n            text = info + message\n            end = \"\\n\"\n        else:\n            info = f\"\\r{self.prefix}{iteration}/{self.total} [{bar}] {percent}% | ET: {ET}| ETR : {ETR} | \"\n            text = info + message\n            end = \"\\r\"\n        print(text, end=end, flush=True)\n\n    @staticmethod\n    def format_time(duration_seconds: float) -> str:\n        \"\"\"\n        Format a duration in seconds into a human-readable string.\n\n        Args:\n            duration_seconds (float): Duration in seconds.\n\n        Returns:\n            str: Formatted time string (e.g., \"1m 23s\", \"45.67s\", \"2h 35m\").\n\n        **Example usage:**\n\n        .. code-block:: python\n\n            formatted_time = ProgressBar.format_time(65.5)  # Returns \"1:05\"\n\n        \"\"\"\n\n        if duration_seconds < 1:\n            formated_time = f\"{duration_seconds * 1e3:.0f}ms\"\n        elif duration_seconds < 60:\n            formated_time = f\"{duration_seconds:.2f}s\"\n        elif duration_seconds < 3600:\n            minutes = int(duration_seconds // 60)\n            seconds = int(duration_seconds % 60)\n            formated_time = f\"{minutes:02d}:{seconds:02d}min\"\n        else:\n            hours = int(duration_seconds // 3600)\n            minutes = int((duration_seconds % 3600) // 60)\n            seconds = int(duration_seconds % 60)\n            formated_time = f\"{hours:02d}:{minutes:02d}:{seconds:02d}\"\n        return formated_time\n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:07.783083Z","iopub.execute_input":"2024-04-08T20:29:07.783441Z","iopub.status.idle":"2024-04-08T20:29:07.799426Z","shell.execute_reply.started":"2024-04-08T20:29:07.783410Z","shell.execute_reply":"2024-04-08T20:29:07.798325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp = pd.read_csv(\"/kaggle/input/ben10/sample_submission.csv\")\nlen(samp)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:48.944943Z","iopub.execute_input":"2024-04-08T20:29:48.945871Z","iopub.status.idle":"2024-04-08T20:29:48.966877Z","shell.execute_reply.started":"2024-04-08T20:29:48.945827Z","shell.execute_reply":"2024-04-08T20:29:48.965920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp = pd.read_csv(\"/kaggle/input/ben10/sample_submission.csv\")\npreds = []\n\nno_iter = len(samp)\npb = ProgressBar(total=no_iter, prefix=\"Processing\", bar_length=50)\n\nfor index, row in samp.iterrows():\n    file_name = row['id']\n    file_path = f\"{test_data_dir}{file_name}\"\n    \n    audio, sr = librosa.load(file_path, sr=16_000)\n    pred = pipe(audio)['text']\n    preds.append(pred)\n    \n    pb.update(index + 1, message=f\"Processing item {index + 1}\")\n\nsamp[\"sentence\"] = preds\n\nsamp.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:32:16.768361Z","iopub.execute_input":"2024-04-08T20:32:16.768758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nfrom kaggle_secrets import UserSecretsClient\n\ndef set_kaggle_api():\n    user_secrets = UserSecretsClient()\n    kaggle_key = user_secrets.get_secret(\"KAGGLE_KEY\")\n    kaggle_username = user_secrets.get_secret(\"KAGGLE_USERNAME\")\n    \n    kaggle_dict = dict(username=kaggle_username, key=kaggle_key)\n    kaggle_json = json.dumps(kaggle_dict)\n\n    os.makedirs('/root/.kaggle', exist_ok=True)\n    with open('/root/.kaggle/kaggle.json', 'w') as f:\n        f.write(kaggle_json)\n    os.chmod('/root/.kaggle/kaggle.json', 0o600)\n    \n    return None\n\nset_kaggle_api()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import subprocess\n\ndef submit_to_kaggle(filename,competition,message=\"\"):\n    command=f'kaggle competitions submit -q {competition} -f {filename} -m \"{message}\"'\n    output = subprocess.run(command, shell=True, capture_output=True, text=True)\n    print(output.stdout)\n    print(output.stderr)\n    return None\n\ncompetition_url_suffix = \"ben10\"\nsubmission_filename = \"submission.csv\"\nsubmission_message = \"bangla-asr-whisper-medium-pretrained-1epoch\"\n\nsubmit_to_kaggle(submission_filename,\n                competition_url_suffix,\n                submission_message)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check submission result\n!kaggle competitions submissions ben10","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:59:29.886279Z","iopub.execute_input":"2024-04-19T06:59:29.886737Z","iopub.status.idle":"2024-04-19T06:59:31.215269Z","shell.execute_reply.started":"2024-04-19T06:59:29.886705Z","shell.execute_reply":"2024-04-19T06:59:31.214050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}