{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":73047,"databundleVersionId":8823072,"sourceType":"competition"},{"sourceId":8220836,"sourceType":"datasetVersion","datasetId":4873610},{"sourceId":8221978,"sourceType":"datasetVersion","datasetId":4874366}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nfrom tqdm import tqdm\n\n# Path to the directory containing the audio files\naudio_dir = \"/kaggle/input/ben10/ben10/16_kHz_valid_audio/\"\n\n# Directory to save the separated audio files\nsave_dir = \"dataset_based_on_region_validation/\"\n\n# Get the list of audio files\naudio_files = [file for file in os.listdir(audio_dir) if file.endswith(\".wav\")]\n\n# Use tqdm to iterate through the audio files and show progress\nfor filename in tqdm(audio_files, desc=\"Copying audio files\"):\n    # Extract the region name from the filename\n    region_name = filename.split(\" (\")[0]\n\n    # Create a directory for the region if it doesn't exist in the save directory\n    region_dir = os.path.join(save_dir, region_name)\n    os.makedirs(region_dir, exist_ok=True)\n\n    # Copy the audio file to the corresponding directory in the save directory\n    source_path = os.path.join(audio_dir, filename)\n    destination_path = os.path.join(region_dir, filename)\n    shutil.copy(source_path, destination_path)  # Copy the file\n\nprint(\"Separation completed.\")\n","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:17:23.195457Z","iopub.execute_input":"2025-06-29T12:17:23.195820Z","iopub.status.idle":"2025-06-29T12:17:36.861752Z","shell.execute_reply.started":"2025-06-29T12:17:23.195794Z","shell.execute_reply":"2025-06-29T12:17:36.860906Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install bnunicodenormalizer","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:17:36.863208Z","iopub.execute_input":"2025-06-29T12:17:36.863463Z","iopub.status.idle":"2025-06-29T12:17:46.602362Z","shell.execute_reply.started":"2025-06-29T12:17:36.863442Z","shell.execute_reply":"2025-06-29T12:17:46.601232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Specify the path of the directory you want to create\ndirectory_path_csv = \"/kaggle/working/csv\"\n##directory_path_sylhet = \"/kaggle/working/sylhet\"\n\n# Create the directory\nos.makedirs(directory_path_csv, exist_ok=True)\n##os.makedirs(directory_path_sylhet, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:17:46.603896Z","iopub.execute_input":"2025-06-29T12:17:46.604175Z","iopub.status.idle":"2025-06-29T12:17:46.608775Z","shell.execute_reply.started":"2025-06-29T12:17:46.604152Z","shell.execute_reply":"2025-06-29T12:17:46.607858Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CHUNK_LENGTH_S = 20.1\nENABLE_BEAM = False\n# none alone 0.9 0.037964275588318684\n# none alone 0.7 0.03592288063510065\n# none alone 0.4 0.03416501275871846\nPUNCT_WEIGHTS = [[1.0, 1.4, 1.0, 0.8]]\n\nif ENABLE_BEAM:\n    BATCH_SIZE = 64\nelse:\n    BATCH_SIZE = 64","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:17:46.611264Z","iopub.execute_input":"2025-06-29T12:17:46.611989Z","iopub.status.idle":"2025-06-29T12:17:46.620664Z","shell.execute_reply.started":"2025-06-29T12:17:46.611969Z","shell.execute_reply":"2025-06-29T12:17:46.619871Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nimport os\nimport librosa\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer,\n    TrainerCallback,\n    TrainingArguments,\n    TrainerState,\n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline\n)\n\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:17:46.621532Z","iopub.execute_input":"2025-06-29T12:17:46.621792Z","iopub.status.idle":"2025-06-29T12:18:03.291379Z","shell.execute_reply.started":"2025-06-29T12:17:46.621774Z","shell.execute_reply":"2025-06-29T12:18:03.290456Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to clean RAM & vRAM\nimport gc\nimport ctypes\n\ndef clean_memory():\n    gc.collect()\n    ctypes.CDLL(\"libc.so.6\").malloc_trim(0)\n    torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:18:03.292538Z","iopub.execute_input":"2025-06-29T12:18:03.293307Z","iopub.status.idle":"2025-06-29T12:18:03.298323Z","shell.execute_reply.started":"2025-06-29T12:18:03.293271Z","shell.execute_reply":"2025-06-29T12:18:03.297386Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pretty_sort(filename):\n    name, number_str = filename.split(\" (\")\n    number = int(number_str.split(\")\")[0])\n    return name, number","metadata":{"execution":{"iopub.status.busy":"2025-06-29T12:18:03.299678Z","iopub.execute_input":"2025-06-29T12:18:03.299985Z","iopub.status.idle":"2025-06-29T12:18:03.312386Z","shell.execute_reply.started":"2025-06-29T12:18:03.299957Z","shell.execute_reply":"2025-06-29T12:18:03.311541Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"/kaggle/input/jiban-ananda/region_based/train_1_no_pretrained_25_epochs_barishal","metadata":{}},{"cell_type":"code","source":"REGION_NAME = ['sylhet','barishal','chittagong','habiganj','kishoreganj','narail','narsingdi','rangpur','sandwip','tangail']\n#sREGION_NAME = ['barishal','chittagong','habiganj','kishoreganj','narail','narsingdi','rangpur','sandwip','tangail']\n\nfor region_name_ in REGION_NAME:\n    MODEL_NAME = f\"/kaggle/input/jiban-ananda/region_based/train_1_no_pretrained_25_epochs_{region_name_}/\"\n\n    print(MODEL_NAME)\n    \n    pipe = pipeline(task=\"automatic-speech-recognition\",\n                    model=MODEL_NAME,\n                    tokenizer=MODEL_NAME,\n                    chunk_length_s=CHUNK_LENGTH_S, device=0, batch_size=BATCH_SIZE)\n    pipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n    print(f\"model loaded!\")\n    \n    BASE_DIR = 'dataset_based_on_region_validation/'\n    test_data_dir = f\"{BASE_DIR}/valid_{region_name_}/\"\n\n    ids = []\n    preds = []\n\n    for root, dirs, files in os.walk(f\"{BASE_DIR}/valid_{region_name_}/\"):\n        files = sorted(files, key=pretty_sort)\n        \n    #     print(files.index(\"valid_sandwip (1).wav\"))\n    #     print(files.index(\"valid_sandwip (132).wav\"))\n        \n    #     put swandip first\n        shift = files[1070 : 1202]\n        \n        files = shift + files[:1070] + files[1202:]\n        ids = files.copy()\n        \n        for file in files:\n            composed_path = f\"{test_data_dir}{file}\"\n            audio, sr = librosa.load(composed_path, sr=16_000)\n            text = pipe(audio)[\"text\"]\n            preds.append(text)\n            print(len(preds))\n            print(preds[len(preds)-1])\n\n        \n        sub_df = pd.DataFrame()\n\n        sub_df[\"id\"] = ids\n        sub_df[\"sentence\"] = preds\n\n        # if region_name_ == 'sylhet':\n        #     sub_df.to_csv(f\"sylhet/submission_{region_name_}.csv\", index=False)\n        # else:\n        #     sub_df.to_csv(f\"csv/submission_{region_name_}.csv\", index=False)\n        sub_df.to_csv(f\"csv/submission_{region_name_}.csv\", index=False)\n        \n        clean_memory()\n            \n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CSV MERGER","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path to the directory containing the CSV files\ndirectory = \"csv/\"\n\n# List to store the DataFrames\ndfs = []\n\n# Iterate through each file in the directory\nfor filename in os.listdir(directory):\n    if filename.endswith(\".csv\"):\n        # Read the CSV file and append its DataFrame to the list\n        df = pd.read_csv(os.path.join(directory, filename))\n        dfs.append(df)\n\n# Concatenate all DataFrames in the list\nmerged_df = pd.concat(dfs, ignore_index=True)\n\n# Save the merged DataFrame to a new CSV file\nmerged_csv_file = \"submission_all.csv\"\nmerged_df.to_csv(merged_csv_file, index=False)\n\nprint(f\"Merged CSV file saved as {merged_csv_file}.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:31.580165Z","iopub.execute_input":"2025-06-29T16:41:31.580436Z","iopub.status.idle":"2025-06-29T16:41:31.644784Z","shell.execute_reply.started":"2025-06-29T16:41:31.580415Z","shell.execute_reply":"2025-06-29T16:41:31.643973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PUNCTUATIONS","metadata":{}},{"cell_type":"code","source":"from transformers import pipeline, AutoModelForTokenClassification, AutoTokenizer\nimport csv\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:31.645885Z","iopub.execute_input":"2025-06-29T16:41:31.646481Z","iopub.status.idle":"2025-06-29T16:41:31.650488Z","shell.execute_reply.started":"2025-06-29T16:41:31.646450Z","shell.execute_reply":"2025-06-29T16:41:31.649719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PUNCT_MODELS = [\n    '/kaggle/input/punctuation-models-jibanananda-das/Punctuation_Models/punct-model-6layers/',\n    '/kaggle/input/punctuation-models-jibanananda-das/Punctuation_Models/punct-model-8layers/',\n    '/kaggle/input/punctuation-models-jibanananda-das/Punctuation_Models/punct-model-11layers/',\n    '/kaggle/input/punctuation-models-jibanananda-das/Punctuation_Models/punct-model-12layers/'\n]\n\nPUNCT_WEIGHTS = [[1.0, 1.4, 1.0, 0.8]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:31.652523Z","iopub.execute_input":"2025-06-29T16:41:31.653261Z","iopub.status.idle":"2025-06-29T16:41:31.661737Z","shell.execute_reply.started":"2025-06-29T16:41:31.653230Z","shell.execute_reply":"2025-06-29T16:41:31.660960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#del pipe\nimport torch\nmodels = [\n    AutoModelForTokenClassification.from_pretrained(f).eval().cuda() for f in PUNCT_MODELS\n]\ntokenizer = AutoTokenizer.from_pretrained(PUNCT_MODELS[0])\ndef punctuate(text):\n    input_ids = tokenizer(text).input_ids\n    with torch.no_grad():\n        model = models[0]\n        logits = torch.nn.functional.softmax(\n            model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n            dim=1).cpu()\n        for model in models[1:]:\n            logits += torch.nn.functional.softmax(\n                model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n                dim=1).cpu()\n        logits = logits / len(models)\n        logits *= torch.FloatTensor(PUNCT_WEIGHTS)\n        label_ids = torch.argmax(logits, dim=-1)\n\n        tokens = tokenizer(text, add_special_tokens=False).input_ids\n        punct_text = \"\"\n        for index, token in enumerate(tokens):\n            token_str = tokenizer.decode(token)\n            if '##' not in token_str:\n                punct_text += \" \" + token_str\n                pass\n            else:\n                punct_text += token_str[2:]\n            punct_text += ['', '।', ',', '?'][label_ids[index].item()]\n\n    punct_text = punct_text.strip()\n    return punct_text\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:31.662545Z","iopub.execute_input":"2025-06-29T16:41:31.662823Z","iopub.status.idle":"2025-06-29T16:41:55.658009Z","shell.execute_reply.started":"2025-06-29T16:41:31.662804Z","shell.execute_reply":"2025-06-29T16:41:55.657025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#del pipe\nimport torch\nmodels = [\n    AutoModelForTokenClassification.from_pretrained(f).eval().cuda() for f in PUNCT_MODELS\n]\ntokenizer = AutoTokenizer.from_pretrained(PUNCT_MODELS[0])\ndef punctuate(text):\n    input_ids = tokenizer(text).input_ids\n    with torch.no_grad():\n        model = models[0]\n        logits = torch.nn.functional.softmax(\n            model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n            dim=1).cpu()\n        for model in models[1:]:\n            logits += torch.nn.functional.softmax(\n                model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n                dim=1).cpu()\n        logits = logits / len(models)\n        logits *= torch.FloatTensor(PUNCT_WEIGHTS)\n        label_ids = torch.argmax(logits, dim=-1)\n\n        tokens = tokenizer(text, add_special_tokens=False).input_ids\n        punct_text = \"\"\n        for index, token in enumerate(tokens):\n            token_str = tokenizer.decode(token)\n            if '##' not in token_str:\n                punct_text += \" \" + token_str\n                pass\n            else:\n                punct_text += token_str[2:]\n            punct_text += ['', '।', ',', '?'][label_ids[index].item()]\n\n    punct_text = punct_text.strip()\n    return punct_text\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the CSV file into a DataFrame\ndf = pd.read_csv(\"submission_all.csv\")\n\n# Apply the punctuate function to each sentence in the 'sentence' column\n#df['sentence'] = df['sentence'].apply(fix_repetition)\ndf['sentence'] = df['sentence'].apply(punctuate)\n\n# Save the modified DataFrame to a new CSV file\ndf.to_csv(\"submission_all_with_puntuations.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:55.659161Z","iopub.execute_input":"2025-06-29T16:41:55.659420Z","iopub.status.idle":"2025-06-29T16:42:43.860722Z","shell.execute_reply.started":"2025-06-29T16:41:55.659399Z","shell.execute_reply":"2025-06-29T16:42:43.859805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ncsv_file = \"submission_all_with_puntuations.csv\"  # Replace with the actual path to your CSV file\ndf = pd.read_csv(csv_file)\n\n# Define the punctuation marks\npunctuation_marks = '!*+,-:;_`>'\n\n# Remove space before punctuation marks in the 'sentence' column\ndf['sentence'] = df['sentence'].str.replace('< >', '<>')\ndf['sentence'] = df['sentence'].str.replace(' !', '!')\ndf['sentence'] = df['sentence'].str.replace(' =', '=')\ndf['sentence'] = df['sentence'].str.replace(' +', '+')\ndf['sentence'] = df['sentence'].str.replace(' *', '*')\ndf['sentence'] = df['sentence'].str.replace(' ,', ',')\ndf['sentence'] = df['sentence'].str.replace(' ;', ';')\ndf['sentence'] = df['sentence'].str.replace(' -', '-')\ndf['sentence'] = df['sentence'].str.replace('- ', '-')\ndf['sentence'] = df['sentence'].str.replace(' <> <> ', ' <> ')\n\n\n\n\n# Save the updated DataFrame back to the same CSV file\ndf.to_csv(csv_file, index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:42:43.861905Z","iopub.execute_input":"2025-06-29T16:42:43.862224Z","iopub.status.idle":"2025-06-29T16:42:43.919934Z","shell.execute_reply.started":"2025-06-29T16:42:43.862197Z","shell.execute_reply":"2025-06-29T16:42:43.919228Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BN_UNI_NORMALIZER","metadata":{}},{"cell_type":"code","source":"# Import\nfrom bnunicodenormalizer import Normalizer \nimport pandas as pd\n\n# Initialize\nnorm = Normalizer(allow_english=True)\n\n# Read the CSV file into a DataFrame\nsub_df = pd.read_csv('submission_all_with_puntuations.csv')\n\n# Define a function to extract the normalized string from the dictionary\ndef extract_normalized_word(word_dict):\n    return word_dict['normalized']\n\n# Iterate over the rows of the DataFrame\nfor index, row in sub_df.iterrows():\n    # Split the sentence into words\n    words = row['sentence'].split()\n    \n    # Normalize each word\n    normalized_words = []\n    for word in words:\n        # Normalize the word and handle NoneType\n        normalized_word = extract_normalized_word(norm(word))\n        if normalized_word:\n            normalized_words.append(normalized_word)\n        else:\n            normalized_words.append(word)  # Use the original word if normalization fails\n    \n    # Join the normalized words back into a sentence\n    normalized_sentence = ' '.join(normalized_words)\n    \n    # Update the 'sentence' column with the normalized sentence\n    sub_df.at[index, 'sentence'] = normalized_sentence\n\n# Save the modified DataFrame back to a CSV file\nsub_df.to_csv('submission_all_with_puntuations_bnuninormalized.csv', index=False)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:42:43.920950Z","iopub.execute_input":"2025-06-29T16:42:43.921290Z","iopub.status.idle":"2025-06-29T16:43:01.757294Z","shell.execute_reply.started":"2025-06-29T16:42:43.921257Z","shell.execute_reply":"2025-06-29T16:43:01.756615Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MERGE SYLHET","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Read both CSV files\ndf1 = pd.read_csv(\"sylhet/submission_sylhet.csv\")\ndf2 = pd.read_csv(\"submission_all_with_puntuations_bnuninormalized.csv\")\n\n# Concatenate the two dataframes vertically\nmerged_df = pd.concat([df1, df2], ignore_index=True)\n\n# Write the merged dataframe to a new CSV file\nmerged_df.to_csv(\"FINAL_submission_all_with_puntuations_bnuninormalized.csv\", index=False)\n\nprint(\"Merging completed.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T18:14:03.716921Z","iopub.execute_input":"2025-06-28T18:14:03.717311Z","iopub.status.idle":"2025-06-28T18:14:03.757156Z","shell.execute_reply.started":"2025-06-28T18:14:03.717277Z","shell.execute_reply":"2025-06-28T18:14:03.756152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T17:19:31.373470Z","iopub.execute_input":"2025-06-29T17:19:31.374224Z","iopub.status.idle":"2025-06-29T17:19:31.388030Z","shell.execute_reply.started":"2025-06-29T17:19:31.374193Z","shell.execute_reply":"2025-06-29T17:19:31.387146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}