{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8823072,"sourceType":"competition"},{"sourceId":4143520,"sourceType":"datasetVersion","datasetId":2447262},{"sourceId":6707460,"sourceType":"datasetVersion","datasetId":3865741},{"sourceId":6833907,"sourceType":"datasetVersion","datasetId":3928977},{"sourceId":6835207,"sourceType":"datasetVersion","datasetId":3929755},{"sourceId":11786214,"sourceType":"datasetVersion","datasetId":7400229}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing libraries","metadata":{"id":"a-Nb01r0LXTB"}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport librosa\nfrom tqdm import tqdm\nimport numpy as np\nimport random\nfrom pydub import AudioSegment\nimport librosa\nimport matplotlib.pyplot as plt\nimport os\nimport librosa\nfrom multiprocessing import Pool\nimport time\nimport tensorflow as tf\nfrom tensorflow.keras.layers import TextVectorization","metadata":{"execution":{"iopub.status.busy":"2025-05-16T16:17:44.803618Z","iopub.execute_input":"2025-05-16T16:17:44.804010Z","iopub.status.idle":"2025-05-16T16:17:55.239697Z","shell.execute_reply.started":"2025-05-16T16:17:44.803978Z","shell.execute_reply":"2025-05-16T16:17:55.238593Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input/ben10/ben10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T17:58:48.719790Z","iopub.execute_input":"2025-05-12T17:58:48.720331Z","iopub.status.idle":"2025-05-12T17:58:49.789137Z","shell.execute_reply.started":"2025-05-12T17:58:48.720306Z","shell.execute_reply":"2025-05-12T17:58:49.788029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/ben10/ben10/16_kHz_train_audio/train.csv\")\ntrain_dir = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/\"\nprint(\"Train dataframe : \")\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:17:57.072119Z","iopub.execute_input":"2025-05-16T16:17:57.072758Z","iopub.status.idle":"2025-05-16T16:17:57.380763Z","shell.execute_reply.started":"2025-05-16T16:17:57.072727Z","shell.execute_reply":"2025-05-16T16:17:57.379531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = \"/kaggle/input/ben10/ben10/16_kHz_valid_audio/\"\ntest_paths = [test_dir+path for path in os.listdir(test_dir)]\ntest = pd.DataFrame(test_paths,columns=['region'])\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:18:08.779009Z","iopub.execute_input":"2025-05-16T16:18:08.779415Z","iopub.status.idle":"2025-05-16T16:18:08.817582Z","shell.execute_reply.started":"2025-05-16T16:18:08.779384Z","shell.execute_reply":"2025-05-16T16:18:08.816264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_regions(path):\n    unwanted_strs = [\"train_\",\"valid_\",\".wav\",\"1\",\"2\",\"3\",\"4\",\"5\",\"6\",\"7\",\"8\",\"9\",\"0\",\"(\",\")\",\" \",\"/kaggle/input/ben/ben/_kHz_audio/\"]\n    for i in unwanted_strs:\n        path = path.replace(i,\"\")\n    return path\n    \n\ndf[\"region\"] = df[\"file_name\"].apply(lambda x:extract_regions(x))\ntest[\"region\"] = test[\"region\"].apply(lambda x:extract_regions(x))\n\n#Plot the distributions\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\ndf.region.value_counts().sort_values().plot(kind='barh', ax=axes[0])\naxes[0].set_title('Distribution of Regions in training set')\n\ntest.region.value_counts().sort_values().plot(kind='barh', ax=axes[1])\naxes[1].set_title('Distribution of Regions in test')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:18:16.197303Z","iopub.execute_input":"2025-05-16T16:18:16.198003Z","iopub.status.idle":"2025-05-16T16:18:16.844555Z","shell.execute_reply.started":"2025-05-16T16:18:16.197970Z","shell.execute_reply":"2025-05-16T16:18:16.843494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(df[\"region\"].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T17:59:52.363049Z","iopub.execute_input":"2025-05-12T17:59:52.363951Z","iopub.status.idle":"2025-05-12T17:59:52.373179Z","shell.execute_reply.started":"2025-05-12T17:59:52.363911Z","shell.execute_reply":"2025-05-12T17:59:52.372112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_paths","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(AudioSegment.from_file('/kaggle/input/ben10/ben10/16_kHz_valid_audio/valid_habiganj (66).wav'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:31:41.919363Z","iopub.execute_input":"2025-05-16T16:31:41.920305Z","iopub.status.idle":"2025-05-16T16:31:42.102946Z","shell.execute_reply.started":"2025-05-16T16:31:41.920270Z","shell.execute_reply":"2025-05-16T16:31:42.101677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"regions = ['barishal', 'chittagong', 'habiganj', 'kishoreganj', 'narail',\n           'narsingdi', 'rangpur', 'sandwip', 'sylhet', 'tangail']\n\nfor region in regions:\n    sample = df[df[\"region\"] == region]\n    \n    if sample.empty:\n        print(f\"No data found for region: {region}\")\n        continue\n\n    idx = random.randint(0, len(sample) - 1)\n\n    file = sample['file_name'].iloc[idx]\n    path = train_dir + file\n\n    print(\"Region :\", region)\n    display(AudioSegment.from_file(path))\n    print(\"Original transcription :\", sample['transcriptions'].iloc[idx])\n    print(\"=\" * 40)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:03:33.737071Z","iopub.execute_input":"2025-05-12T18:03:33.737747Z","iopub.status.idle":"2025-05-12T18:03:36.349805Z","shell.execute_reply.started":"2025-05-12T18:03:33.737716Z","shell.execute_reply":"2025-05-12T18:03:36.348923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp /kaggle/input/bengali-eval-data/predict.py .\n\n!cp -r ../input/python-packages2 ./\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/python-Levenshtein-0.12.2.tar.gz -f ./ --no-index\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:04:15.584658Z","iopub.execute_input":"2025-05-12T18:04:15.585421Z","iopub.status.idle":"2025-05-12T18:04:40.803545Z","shell.execute_reply.started":"2025-05-12T18:04:15.585390Z","shell.execute_reply":"2025-05-12T18:04:40.802533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport csv\nimport time\nimport glob\n\nMODEL = '/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium/'\n\nCHUNK_LENGTH_S = 20.1\nENABLE_BEAM = True\n\nif ENABLE_BEAM:\n    BATCH_SIZE = 4\nelse:\n    BATCH_SIZE = 8\nfrom transformers import pipeline\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:04:53.049545Z","iopub.execute_input":"2025-05-12T18:04:53.050477Z","iopub.status.idle":"2025-05-12T18:04:59.522408Z","shell.execute_reply.started":"2025-05-12T18:04:53.050443Z","shell.execute_reply":"2025-05-12T18:04:59.521494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe = pipeline(task=\"automatic-speech-recognition\",\n                model=MODEL,\n                tokenizer=MODEL,\n                chunk_length_s=CHUNK_LENGTH_S,device=0, batch_size=BATCH_SIZE)\npipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\nprint(\"model loaded!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:05:02.774553Z","iopub.execute_input":"2025-05-12T18:05:02.774926Z","iopub.status.idle":"2025-05-12T18:05:30.008928Z","shell.execute_reply.started":"2025-05-12T18:05:02.774896Z","shell.execute_reply":"2025-05-12T18:05:30.008062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if ENABLE_BEAM:\n    texts = pipe(test_paths, generate_kwargs={\"max_length\": 260, \"num_beams\": 4})\nelse:\n    texts = pipe(test_paths)\npreds = []\nfor i in texts:\n    preds.append(i['text'])\nlen(preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:05:34.499260Z","iopub.execute_input":"2025-05-12T18:05:34.499608Z","iopub.status.idle":"2025-05-12T20:24:10.653004Z","shell.execute_reply.started":"2025-05-12T18:05:34.499580Z","shell.execute_reply":"2025-05-12T20:24:10.652037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.DataFrame({\"id\":test_paths,\"sentence\":preds})\nsub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T20:24:10.654609Z","iopub.execute_input":"2025-05-12T20:24:10.654886Z","iopub.status.idle":"2025-05-12T20:24:10.665094Z","shell.execute_reply.started":"2025-05-12T20:24:10.654863Z","shell.execute_reply":"2025-05-12T20:24:10.664094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub['id'] = sub['id'].apply(lambda x: x.replace(test_dir,\"\"))\nsub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T20:24:10.666638Z","iopub.execute_input":"2025-05-12T20:24:10.666953Z","iopub.status.idle":"2025-05-12T20:24:10.686854Z","shell.execute_reply.started":"2025-05-12T20:24:10.666924Z","shell.execute_reply":"2025-05-12T20:24:10.686009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.to_csv(\"submission1.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T20:31:32.769467Z","iopub.execute_input":"2025-05-12T20:31:32.769887Z","iopub.status.idle":"2025-05-12T20:31:32.789870Z","shell.execute_reply.started":"2025-05-12T20:31:32.769856Z","shell.execute_reply":"2025-05-12T20:31:32.788872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading model and setting up vocabulary","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/bangla-text2ipa-transformer-model/text2ipa-transformer-model'\nnew_model=tf.saved_model.load(path)\n\nvb = ['', '[UNK]', '[start]', '[end]', 'া', 'র', '্', 'ে', 'ি', 'ন', 'ক', 'ব', 'স', 'ল', 'ত', 'ম', 'প', 'ু', 'দ', 'ট', 'য়', 'জ', '।', 'ো', 'গ', 'হ', 'য', 'শ', 'ী', 'ই', 'চ', 'ভ', 'আ', 'ও', 'ছ', 'ষ', 'ড', 'ফ', 'অ', 'ধ', 'খ', 'ড়', 'উ', 'ণ', 'এ', 'থ', 'ং', 'ঁ', 'ূ', 'ৃ', 'ঠ', 'ঘ', 'ঞ', 'ঙ', 'ৌ', '‘', 'ৎ', 'ঝ', 'ৈ', '়', 'ঢ', 'ঃ', 'ঈ', '\\u200c', 'ৗ', 'a', 'ঐ', 'd', 'w', 'ঋ', 'i', 'e', 't', 's', 'n', 'm', 'b', '“', 'u', 'r', 'œ', 'o', '–', 'ঊ', 'ঢ়', 'Í', 'g', 'p', '\\xad', 'h', 'c', 'l', 'ঔ', 'ƒ', '”', 'Ñ', '¡', 'y', 'j', 'f', '→', '—', 'ø', 'è', '¦', '¥', 'x', 'v', 'k']\nvipa = ['', '[UNK]', '[start]', '[end]', 'ɐ', 'ɾ', 'i', 'o', 'e', '̪', 't', 'n', 'k', 'ɔ', 'ʃ', 'b', 'd', 'l', 'u', 'p', 'm', 'ʰ', 'ɟ', '͡', '̯', 'g', 'ʱ', '।', 'c', 'ʲ', 'h', 's', 'ŋ', 'ɛ', 'ɽ', '̃', 'ʷ', '‘', '“', '–', '”', '—', 'w', 'j']\nv = vb + vipa\ns = set()\nfor ch in v:\n  s.add(ch)\n\nvocab = sorted(list(s))\nprint(\"Length of vocab:\", len(s))\nprint(vocab)\nvocab_size = len(vocab)\n\nsequence_length = 64 # 20\nbatch_size = 64\n\neng_vectorization = TextVectorization(\n    max_tokens=vocab_size, output_mode=\"int\", output_sequence_length=sequence_length,\n    vocabulary=vocab\n)\n\nspa_vectorization = TextVectorization(\n    max_tokens=vocab_size,\n    output_mode=\"int\",\n    output_sequence_length=sequence_length + 1,\n    vocabulary=vocab\n)\n\nspa_vocab = spa_vectorization.get_vocabulary()\nspa_index_lookup = dict(zip(range(len(spa_vocab)), spa_vocab))\nmax_decoded_sentence_length = 64 #20\n\ndef decode_sequence(input_sentence):\n    tokenized_input_sentence = eng_vectorization([input_sentence])\n    decoded_sentence = '[start]'\n\n    for i in range(max_decoded_sentence_length):\n        tokenized_target_sentence = spa_vectorization([decoded_sentence])[:, :-1]\n        predictions = new_model([tokenized_input_sentence, tokenized_target_sentence])\n        sampled_token_index = np.argmax(predictions[0, i, :])\n        sampled_token = spa_index_lookup[sampled_token_index]\n        decoded_sentence += \" \" + sampled_token\n        if sampled_token == '[UNK]':\n            break\n    return decoded_sentence\n\ndef sentence_word(sentence):\n  trg=''\n  for ch in sentence:\n      if ch != \" \":\n        trg += ch\n  return trg\n\ndef word_sentence(word):\n  sentence = \"\"\n  for ch in word:\n    sentence += (ch + \" \")\n  return sentence\n","metadata":{"id":"CO4-b5G-Gz1V","outputId":"81a2ee5e-269c-422d-c2d7-c9e3c27b9484","execution":{"iopub.status.busy":"2025-05-12T20:31:38.287010Z","iopub.execute_input":"2025-05-12T20:31:38.287391Z","iopub.status.idle":"2025-05-12T20:31:41.522789Z","shell.execute_reply.started":"2025-05-12T20:31:38.287360Z","shell.execute_reply":"2025-05-12T20:31:41.521848Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing steps","metadata":{"id":"5ARtfO3XGz1W"}},{"cell_type":"code","source":"def bangla_vocabulary():\n  Vowels = ['অ', 'আ', 'ই', 'ঈ', 'উ', 'ঊ', 'ঋ', 'ঌ', 'এ', 'ঐ', 'ও', 'ঔ']\n  Vowel_signs = ['া', 'ি', 'ী', 'ু', 'ূ', 'ৃ', 'ৄ', 'ে', 'ৈ', 'ো', 'ৌ']\n  Consonants = ['ক', 'খ', 'গ', 'ঘ', 'ঙ', 'চ', 'ছ', 'জ', 'ঝ', 'ঞ', 'ট', 'ঠ', 'ড', 'ঢ', 'ণ', 'ত', 'থ', 'দ', 'ধ', 'ন', 'প', 'ফ', 'ব', 'ভ', 'ম', 'য', 'র', 'ল', 'শ', 'ষ', 'স', 'হ', 'ড়', 'ঢ়', 'য়', 'ৎ', 'ং', 'ঃ', 'ঁ']\n  Operators = ['=', '+', '-', '*', '/', '%', '<', '>', '×', '÷']\n  Punctuation_marks = ['।', ',', ';', ':', '?', '!', \"'\", '.', '\"', '-', '[', ']', '{', '}', '(', ')', '–', '—', '―', '~']\n  Others = ['্', '়', 'ৗ', '‘', '’', '“', '”']\n\n  BANGLA_VOCAB = sorted(list(set(Vowels + Vowel_signs + Consonants +  Operators + Punctuation_marks + Others)))\n  return BANGLA_VOCAB\n\ndef foreign_character_normalization(word):\n  BANGLA_VOCAB = bangla_vocabulary()\n  normalized_word = \"\"\n\n  for ch in word:\n    if ch not in BANGLA_VOCAB:\n      continue\n    normalized_word += ch\n  return normalized_word\n\ndef aligned_stateful_tokenizer(word):\n  vocab = [ 'ঁ', 'ং', 'ঃ', 'অ', 'আ', 'ই', 'ঈ', 'উ', 'ঊ', 'ঋ', 'এ', 'ঐ', 'ও', 'ঔ', 'ক', 'খ', 'গ', 'ঘ', 'ঙ', 'চ', 'ছ', 'জ', 'ঝ', 'ঞ', 'ট', 'ঠ', 'ড', 'ঢ', 'ণ', 'ত', 'থ', 'দ', 'ধ', 'ন', 'প', 'ফ', 'ব', 'ভ', 'ম', 'য', 'র', 'ল', 'শ', 'ষ', 'স', 'হ', '়', 'া', 'ি', 'ী', 'ু', 'ূ', 'ৃ', 'ে', 'ৈ', 'ো', 'ৌ', '্', 'ৎ', 'ৗ', 'ড়', 'ঢ়', 'য়']\n  n = len(word)\n  i = 0\n  j = n-1\n\n  state = []\n  tokens = []\n\n  while i < n:\n    subword = \"\"\n    if word[i] in vocab:\n      found = True\n      while i < n and word[i] in vocab:\n        subword += word[i]\n        i += 1\n\n    elif not(word[i] in vocab):\n      found = False\n      while i < n and not(word[i] in vocab):\n        subword += word[i]\n        i += 1\n\n    state.append(found)\n    tokens.append(subword)\n  return state, tokens\n\ndef preprocess(word):\n  preprocessed_word = foreign_character_normalization(word)\n  return preprocessed_word","metadata":{"id":"Sev_weRUGz1W","execution":{"iopub.status.busy":"2025-05-12T20:31:47.454548Z","iopub.execute_input":"2025-05-12T20:31:47.454882Z","iopub.status.idle":"2025-05-12T20:31:47.465708Z","shell.execute_reply.started":"2025-05-12T20:31:47.454853Z","shell.execute_reply":"2025-05-12T20:31:47.464816Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading Dictionary from Training set","metadata":{"id":"NJl8tpVlGz1Z"}},{"cell_type":"code","source":"path = \"/kaggle/input/text2ipa-mapping-trainset/previous_trainset_word_ipa_map_37807.csv\"\ndf = pd.read_csv(path)\n\nDICTIONARY = {}\nvocab = ['a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z']\ndigits = ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9']\nENGLISH_VOCAB = vocab + digits\n\nproblem = []\nfor index, row in df.iterrows():\n  word = row['word']\n  ipa = row['ipa']\n  DICTIONARY[word] = ipa\n\n# correcting incorrect annotaions\nDICTIONARY[\"seen\"] = \"\"\nDICTIONARY[\"passage\"] = \"\"\nDICTIONARY[\"Writing\"] = \"\"\nDICTIONARY[\"Test\"] = \"\"\nDICTIONARY[\"B\"] = \"\"\nDICTIONARY[\"admissions\"] = \"\"\n\nprint(\"Total train data\", len(DICTIONARY))","metadata":{"id":"Hewa9fMXGz1a","outputId":"b2fe326e-374a-4bd6-97dc-39ec45fd85a8","execution":{"iopub.status.busy":"2025-05-12T20:31:57.126391Z","iopub.execute_input":"2025-05-12T20:31:57.126727Z","iopub.status.idle":"2025-05-12T20:31:59.060293Z","shell.execute_reply.started":"2025-05-12T20:31:57.126701Z","shell.execute_reply":"2025-05-12T20:31:59.059039Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nurl_test = \"/kaggle/input/submission-csv/submission.csv\"\ndf_test = pd.read_csv(url_test)\nprint(\"Shape:\", df_test.shape)\ndf_test.head(2)","metadata":{"id":"E81dAgMkKzIX","outputId":"471ce64b-c641-4a07-ce00-0b333119779d","execution":{"iopub.status.busy":"2025-05-12T20:32:04.585870Z","iopub.execute_input":"2025-05-12T20:32:04.586226Z","iopub.status.idle":"2025-05-12T20:32:04.625595Z","shell.execute_reply.started":"2025-05-12T20:32:04.586184Z","shell.execute_reply":"2025-05-12T20:32:04.624474Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for index, row in df_test.iterrows():\n  row_id = row['id']\n  text = row['sentence']\n  texts = text.split()\n\n  for word in texts:\n    if word in DICTIONARY.keys():\n      continue\n\n    normalized_word = foreign_character_normalization(word)\n    state, tokens = aligned_stateful_tokenizer(normalized_word)\n\n    if len(normalized_word) == 0:\n      DICTIONARY[word] = \"\"\n      continue\n\n    for i in range(len(state)):\n      if state[i]:\n        tokenized_word = tokens[i]\n        translated = decode_sequence(word_sentence(tokenized_word))\n        trg = sentence_word(translated)\n        trg = trg[7:]\n        trg = trg[:-5]\n        tokens[i] = trg\n\n    value = \"\".join(tokens)\n    DICTIONARY[word] = value\n\n  print(\"----->\", index)","metadata":{"id":"krMsZawBGz1c","outputId":"b3218098-46fb-4f77-b0e1-bedf7c0cbfdc","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(DICTIONARY))","metadata":{"id":"JZfUNyfyGz1d","outputId":"e2c1695f-cae5-4d35-c592-ffccbc28b726","execution":{"iopub.status.busy":"2025-05-12T20:48:32.231273Z","iopub.execute_input":"2025-05-12T20:48:32.231636Z","iopub.status.idle":"2025-05-12T20:48:32.236681Z","shell.execute_reply.started":"2025-05-12T20:48:32.231607Z","shell.execute_reply":"2025-05-12T20:48:32.235693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Post processing","metadata":{"id":"CeXc2SATGz1e"}},{"cell_type":"code","source":"import pandas as pd\n\nBANGLA_DIGIT = ['১', '২', '৩', '৪', '৫', '৬', '৭', '৮', '৯', '০']\n\ndef load_dic(DICTIONARY):\n  dic = {}\n  for key in DICTIONARY:\n    word = key\n    ipa = DICTIONARY[key]\n    if not(type(ipa) == type('cat')):\n      ipa = word\n\n    # Eleminiting bangla digits\n    ipa_ = \"\"\n    for ch in ipa:\n      if ch not in BANGLA_DIGIT:\n        ipa_ += ch\n    dic[word] = ipa_\n\n  print(\"Dictionary Loaded...\")\n  return dic\n\ndic = load_dic(DICTIONARY)","metadata":{"id":"5q0fPxVPGz1e","outputId":"0ca427fe-2376-4c8e-f697-533844d7de34","execution":{"iopub.status.busy":"2025-05-12T20:48:41.068666Z","iopub.execute_input":"2025-05-12T20:48:41.069329Z","iopub.status.idle":"2025-05-12T20:48:41.201076Z","shell.execute_reply.started":"2025-05-12T20:48:41.069298Z","shell.execute_reply":"2025-05-12T20:48:41.200165Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_submission(dic):\n  url_test = \"/kaggle/input/submission-csv/submission.csv\"\n  df_test = pd.read_csv(url_test)\n  rows = []\n  ipas = []\n\n  for index, row in df_test.iterrows():\n    row_id = row['id']\n    text = row['sentence']\n    texts = text.split()\n    pred = []\n\n    for word in texts:\n      ipa = dic[word]\n      pred.append(ipa)\n\n    ipa_text = \" \".join(pred)\n    rows.append(row_id)\n    ipas.append(ipa_text)\n\n  return rows, ipas\nrows, ipas = generate_submission(dic)\nprint(len(rows), len(ipas))","metadata":{"id":"jNEHOyJ4Gz1f","outputId":"1f261c02-a5f1-4d1f-94d9-7b6719a28dec","execution":{"iopub.status.busy":"2025-05-12T20:48:45.214026Z","iopub.execute_input":"2025-05-12T20:48:45.214715Z","iopub.status.idle":"2025-05-12T20:48:45.339296Z","shell.execute_reply.started":"2025-05-12T20:48:45.214685Z","shell.execute_reply":"2025-05-12T20:48:45.338472Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Saving submission file","metadata":{}},{"cell_type":"code","source":"def submission_file(rows, ipas):\n  data = {\n    'row_id_column_name': rows,\n    'ipa': ipas\n  }\n\n  df = pd.DataFrame(data, columns=data.keys())\n  df.to_csv('/kaggle/working/submission2.csv', index=False)\nsubmission_file(rows, ipas)","metadata":{"id":"KcszQ9R2Gz1g","execution":{"iopub.status.busy":"2025-05-12T20:48:48.393638Z","iopub.execute_input":"2025-05-12T20:48:48.394569Z","iopub.status.idle":"2025-05-12T20:48:48.415465Z","shell.execute_reply.started":"2025-05-12T20:48:48.394535Z","shell.execute_reply":"2025-05-12T20:48:48.414595Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1 = pd.read_csv('/kaggle/working/submission2.csv')\ndf1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T20:50:05.561539Z","iopub.execute_input":"2025-05-12T20:50:05.562317Z","iopub.status.idle":"2025-05-12T20:50:05.591868Z","shell.execute_reply.started":"2025-05-12T20:50:05.562275Z","shell.execute_reply":"2025-05-12T20:50:05.591004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T20:50:11.383824Z","iopub.execute_input":"2025-05-12T20:50:11.384170Z","iopub.status.idle":"2025-05-12T20:50:11.395121Z","shell.execute_reply.started":"2025-05-12T20:50:11.384141Z","shell.execute_reply":"2025-05-12T20:50:11.394229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}