{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":8217311,"sourceType":"datasetVersion","datasetId":4847613}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **The Ultimate Post Processing Function**","metadata":{}},{"cell_type":"code","source":"!pip install bnunicodenormalizer\n\nimport re\nfrom bnunicodenormalizer import Normalizer \nimport pandas as pd","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-04-24T16:09:02.602092Z","iopub.execute_input":"2024-04-24T16:09:02.602732Z","iopub.status.idle":"2024-04-24T16:09:23.974056Z","shell.execute_reply.started":"2024-04-24T16:09:02.602695Z","shell.execute_reply":"2024-04-24T16:09:23.972607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punctuation(text):\n    # Remove unnecessary & unusual punc patterns\n    punctSeq = u\"['\\\"“”‘’]+|[…]+|[:;]+\"\n    punc = u\"[()$%^&*+={}\\[\\]:\\\"|\\'\\~`<>/¦½£¶¼©⅐⅑⅒⅓⅔⅕⅖⅗⅘⅙⅚⅛⅜⅝⅞⅟↉¤¿º;-]+\"\n    punc = punc.replace('<>', '')\n    text = re.sub(punctSeq, \" \", text)\n    text = re.sub(punc, \" \", text)\n    # Exclude comma, bengali full stop & question mark from punctuation removal\n    text = re.sub('[!\"#$.@[\\]^_`{|}~]', ' ', text)\n    text = text.replace(\"\\\\\", \" \")\n    return text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:28.099935Z","iopub.execute_input":"2024-04-24T16:09:28.100511Z","iopub.status.idle":"2024-04-24T16:09:28.109400Z","shell.execute_reply.started":"2024-04-24T16:09:28.100473Z","shell.execute_reply":"2024-04-24T16:09:28.107991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_whitespace_before_punctuation(text):\n    # Define the regular expression pattern to match whitespace before comma, question mark, and Bengali full stop\n    pattern = r'(\\S)\\s*([,?।])'\n    # Substitute the matches with just the leading word and the punctuation mark without any whitespace\n    result = re.sub(pattern, r'\\1\\2', text)\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:30.424750Z","iopub.execute_input":"2024-04-24T16:09:30.425173Z","iopub.status.idle":"2024-04-24T16:09:30.430886Z","shell.execute_reply.started":"2024-04-24T16:09:30.425143Z","shell.execute_reply":"2024-04-24T16:09:30.429570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_unk(text):\n    cleaned_text = re.sub(r'UNK', '<>', text)\n    return cleaned_text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:33.695674Z","iopub.execute_input":"2024-04-24T16:09:33.696130Z","iopub.status.idle":"2024-04-24T16:09:33.701800Z","shell.execute_reply.started":"2024-04-24T16:09:33.696088Z","shell.execute_reply":"2024-04-24T16:09:33.700739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_text(text):\n    # 1. Initialize normalizer\n    bnorm = Normalizer(\n        allow_english = False,\n        keep_legacy_symbols = True,\n        legacy_maps = 'default',\n    )\n    \n    # 2. Tokenize the text into words\n    words = text.split()\n    \n    # 3. Normalize each word\n    normalized_words = [bnorm(word)[\"normalized\"] for word in words if isinstance(word, str)]\n    \n    # 4. Reconstruct the text with normalized words\n    normalized_text = ' '.join(word for word in normalized_words if word is not None)\n    \n    return normalized_text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:36.211594Z","iopub.execute_input":"2024-04-24T16:09:36.212929Z","iopub.status.idle":"2024-04-24T16:09:36.221104Z","shell.execute_reply.started":"2024-04-24T16:09:36.212880Z","shell.execute_reply":"2024-04-24T16:09:36.219547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_repeating_chars_in_words(text):\n    words = text.split()\n    cleaned_words = []\n    for word in words:\n        cleaned_word = word[0]\n        for i in range(1, len(word)):\n            if word[i] != word[i - 1]:\n                cleaned_word += word[i]\n        cleaned_words.append(cleaned_word)\n    cleaned_text = ' '.join(cleaned_words)\n    return cleaned_text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:38.867635Z","iopub.execute_input":"2024-04-24T16:09:38.868144Z","iopub.status.idle":"2024-04-24T16:09:38.875685Z","shell.execute_reply.started":"2024-04-24T16:09:38.868109Z","shell.execute_reply":"2024-04-24T16:09:38.874382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_repeating_words(text):\n    \"\"\" we remove repeating words whic repeats more than twice\n        in a sequence \"\"\"\n    words = text.split()\n    if len(words)>0:\n        cleaned_words = [words[0]]  # Initialize with the first word\n    else:\n        cleaned_words = \"\"\n    count = 1  # Initialize count for the first word\n    for i in range(1, len(words)):\n        if words[i] == words[i - 1]:\n            count += 1\n            if count <= 2:  # Keep up to 2 occurrences\n                cleaned_words.append(words[i])\n        else:\n            cleaned_words.append(words[i])\n            count = 1  # Reset count for a new word\n    cleaned_text = ' '.join(cleaned_words)\n    return cleaned_text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:42.474571Z","iopub.execute_input":"2024-04-24T16:09:42.475021Z","iopub.status.idle":"2024-04-24T16:09:42.485438Z","shell.execute_reply.started":"2024-04-24T16:09:42.474987Z","shell.execute_reply":"2024-04-24T16:09:42.484176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_process(text):\n    # 1. Remove leading and trailing white spaces\n    text = text.strip()\n    \n    # 2. Remove unnecessary punctuations\n    text = remove_punctuation(text)\n    \n    # 3. Remove extra white spaces between words\n    text = re.sub(r'\\s+', ' ', text)\n    \n    # 4. Remove extra white spaces between punctuation & words\n    text = remove_whitespace_before_punctuation(text)\n    \n    # 5. Remove None Type eg. [UNK]\n    text = remove_unk(text)\n\n    # 6. Normalize words\n    text = normalize_text(text)\n    \n    # 7. Remove repeating chars in words\n    text = remove_repeating_chars_in_words(text)\n    \n    # 8. Remove repeating words in the text\n    text = remove_repeating_words(text)\n    \n    return text","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:45.058295Z","iopub.execute_input":"2024-04-24T16:09:45.058741Z","iopub.status.idle":"2024-04-24T16:09:45.068092Z","shell.execute_reply.started":"2024-04-24T16:09:45.058707Z","shell.execute_reply":"2024-04-24T16:09:45.066663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now feed all the texts to the \"post_process()\" function","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/submission-file/.77280.csv\")\ndf.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-04-24T16:09:48.400503Z","iopub.execute_input":"2024-04-24T16:09:48.401089Z","iopub.status.idle":"2024-04-24T16:09:48.477026Z","shell.execute_reply.started":"2024-04-24T16:09:48.401047Z","shell.execute_reply":"2024-04-24T16:09:48.475801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['sentence'] = df['sentence'].apply(post_process)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:09:52.882189Z","iopub.execute_input":"2024-04-24T16:09:52.882571Z","iopub.status.idle":"2024-04-24T16:10:28.578529Z","shell.execute_reply.started":"2024-04-24T16:09:52.882539Z","shell.execute_reply":"2024-04-24T16:10:28.577263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:10:40.498948Z","iopub.execute_input":"2024-04-24T16:10:40.499379Z","iopub.status.idle":"2024-04-24T16:10:40.510474Z","shell.execute_reply.started":"2024-04-24T16:10:40.499346Z","shell.execute_reply":"2024-04-24T16:10:40.509116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:10:43.256309Z","iopub.execute_input":"2024-04-24T16:10:43.256725Z","iopub.status.idle":"2024-04-24T16:10:43.291725Z","shell.execute_reply.started":"2024-04-24T16:10:43.256693Z","shell.execute_reply":"2024-04-24T16:10:43.290300Z"},"trusted":true},"execution_count":null,"outputs":[]}]}