{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8076,"databundleVersionId":44219,"sourceType":"competition"},{"sourceId":16295,"databundleVersionId":1099992,"sourceType":"competition"},{"sourceId":17777,"databundleVersionId":869809,"sourceType":"competition"},{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"},{"sourceId":9801,"sourceType":"datasetVersion","datasetId":6763},{"sourceId":12944,"sourceType":"datasetVersion","datasetId":9235},{"sourceId":42887,"sourceType":"datasetVersion","datasetId":32801},{"sourceId":1246668,"sourceType":"datasetVersion","datasetId":715814}],"dockerImageVersionId":29994,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a id=\"Read_and_explore_data\"></a>\n\n# Read and explore data\n\n<a id=\"Importing_Main_Packages\"></a>\n## Importing Main Packages\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"%time\nimport os\nimport sys\nimport warnings\nif not sys.warnoptions:\n    warnings.simplefilter(\"ignore\")\n    \nimport numpy as np\nimport pandas as pd\nimport sklearn\n\n# Libraries and packages for text (pre-)processing \nimport string\nimport re\nimport nltk\nfrom transformers import BertTokenizer\n\ntokenizer = BertTokenizer.from_pretrained(\"bert-large-uncased\")\n\nprint(\"Python version:\", sys.version)\nprint(\"Version info.:\", sys.version_info)\nprint(\"pandas version:\", pd.__version__)\nprint(\"numpy version:\", np.__version__)\nprint(\"skearn version:\", sklearn.__version__)\nprint(\"re version:\", re.__version__)\nprint(\"nltk version:\", nltk.__version__)\n\n!pip install pyicu\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:24.988206Z","iopub.execute_input":"2026-01-16T23:08:24.988528Z","iopub.status.idle":"2026-01-16T23:08:28.521810Z","shell.execute_reply.started":"2026-01-16T23:08:24.988493Z","shell.execute_reply":"2026-01-16T23:08:28.520956Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Read_the_Data\"></a>\n## Read the Data","metadata":{}},{"cell_type":"code","source":"%time\n\n# read the csv file\ntrain_df = pd.read_csv(\"/kaggle/input/nlp-getting-started/train.csv\")\ndisplay(train_df.shape, train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:28.524478Z","iopub.execute_input":"2026-01-16T23:08:28.524877Z","iopub.status.idle":"2026-01-16T23:08:28.560975Z","shell.execute_reply.started":"2026-01-16T23:08:28.524827Z","shell.execute_reply":"2026-01-16T23:08:28.560312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# some early explorations\n\ndisplay(train_df[~train_df[\"location\"].isnull()].head())\ndisplay(train_df[train_df[\"target\"] == 0][\"text\"].values[1])\ndisplay(train_df[train_df[\"target\"] == 1][\"text\"].values[1])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:28.562262Z","iopub.execute_input":"2026-01-16T23:08:28.562507Z","iopub.status.idle":"2026-01-16T23:08:28.580655Z","shell.execute_reply.started":"2026-01-16T23:08:28.562480Z","shell.execute_reply":"2026-01-16T23:08:28.579795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Text_Cleaning\"></a>\n\n# Text Cleaning:\n\n<a id=\"Capitalization\"></a>\n## Capitalization/ Lower case\nThe most common approach in text cleaning is capitalization or lower case due to the diversity of capitalization to form a sentence. This technique will project all words in text and document into the same feature space. However, it would also cause problems with exceptional cases such as the USA or UK, which could be solved by replacing typos, slang, acronyms or informal abbreviations technique.\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"train_df[\"text_clean\"] = train_df[\"text\"].apply(lambda x: x.lower())\ndisplay(train_df.head())","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:28.581919Z","iopub.execute_input":"2026-01-16T23:08:28.582172Z","iopub.status.idle":"2026-01-16T23:08:28.598740Z","shell.execute_reply.started":"2026-01-16T23:08:28.582148Z","shell.execute_reply":"2026-01-16T23:08:28.598019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Expand_the_Contractions\"></a>\n## Expand the Contractions\nWe use the [contractions package](https://github.com/kootenpv/contractions) to expand the contraction in English such as we'll -> we will or we shouldn't've -> we should not have.\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"# Intall the contractions package - https://github.com/kootenpv/contractions\n!pip install contractions","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:28.601123Z","iopub.execute_input":"2026-01-16T23:08:28.601405Z","iopub.status.idle":"2026-01-16T23:08:34.387627Z","shell.execute_reply.started":"2026-01-16T23:08:28.601378Z","shell.execute_reply":"2026-01-16T23:08:34.386846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\nimport contractions\n\n# Test\ntest_text = \"\"\"\n            Y'all can't expand contractions I'd think. I'd like to know how I'd done that! \n            We're going to the zoo and I don't think I'll be home for dinner.\n            Theyre going to the zoo and she'll be home for dinner.\n            We should've do it in here but we shouldn't've eat it\n            \"\"\"\nprint(\"Test: \", contractions.fix(test_text))\n\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: contractions.fix(x))\n\n# double check\nprint(train_df[\"text\"][67])\nprint(train_df[\"text_clean\"][67])\nprint(train_df[\"text\"][12])\nprint(train_df[\"text_clean\"][12])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.390307Z","iopub.execute_input":"2026-01-16T23:08:34.390590Z","iopub.status.idle":"2026-01-16T23:08:34.470307Z","shell.execute_reply.started":"2026-01-16T23:08:34.390555Z","shell.execute_reply":"2026-01-16T23:08:34.469490Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Noise_Removal\"></a>\n\n## Noise Removal \nText data could include various unnecessary characters or punctuation such as URLs, HTML tags, non-ASCII characters, or other special characters (symbols, emojis, and other graphic characters). \n\n<a id=\"Remove_urls\"></a>\n### Remove URLs\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"def remove_URL(text):\n    \"\"\"\n        Remove URLs from a sample string\n    \"\"\"\n    return re.sub(r\"https?://\\S+|www\\.\\S+\", \"\", text)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.471703Z","iopub.execute_input":"2026-01-16T23:08:34.472112Z","iopub.status.idle":"2026-01-16T23:08:34.475968Z","shell.execute_reply.started":"2026-01-16T23:08:34.472082Z","shell.execute_reply":"2026-01-16T23:08:34.475303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove urls from the text\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: remove_URL(x))\n\n# double check\nprint(train_df[\"text\"][31])\nprint(train_df[\"text_clean\"][31])\nprint(train_df[\"text\"][37])\nprint(train_df[\"text_clean\"][37])\nprint(train_df[\"text\"][62])\nprint(train_df[\"text_clean\"][62])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.477151Z","iopub.execute_input":"2026-01-16T23:08:34.477419Z","iopub.status.idle":"2026-01-16T23:08:34.507491Z","shell.execute_reply.started":"2026-01-16T23:08:34.477397Z","shell.execute_reply":"2026-01-16T23:08:34.506751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Remove_HTML_tags\"></a>\n\n### Remove HTML tags\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"def remove_html(text):\n    \"\"\"\n        Remove the html in sample text\n    \"\"\"\n    html = re.compile(r\"<.*?>|&([a-z0-9]+|#[0-9]{1,6}|#x[0-9a-f]{1,6});\")\n    return re.sub(html, \"\", text)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.508990Z","iopub.execute_input":"2026-01-16T23:08:34.509275Z","iopub.status.idle":"2026-01-16T23:08:34.513719Z","shell.execute_reply.started":"2026-01-16T23:08:34.509241Z","shell.execute_reply":"2026-01-16T23:08:34.512899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove html from the text\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: remove_html(x))\n\n# double check\nprint(train_df[\"text\"][62])\nprint(train_df[\"text_clean\"][62])\nprint(train_df[\"text\"][7385])\nprint(train_df[\"text_clean\"][7385])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.515326Z","iopub.execute_input":"2026-01-16T23:08:34.515686Z","iopub.status.idle":"2026-01-16T23:08:34.548981Z","shell.execute_reply.started":"2026-01-16T23:08:34.515649Z","shell.execute_reply":"2026-01-16T23:08:34.548271Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Remove_Non_ASCII\"></a>\n\n### Remove Non-ASCI:\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"def remove_non_ascii(text):\n    \"\"\"\n        Remove non-ASCII characters \n    \"\"\"\n    return re.sub(r'[^\\x00-\\x7f]',r'', text) # or ''.join([x for x in text if x in string.printable]) ","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.550202Z","iopub.execute_input":"2026-01-16T23:08:34.550546Z","iopub.status.idle":"2026-01-16T23:08:34.555185Z","shell.execute_reply.started":"2026-01-16T23:08:34.550511Z","shell.execute_reply":"2026-01-16T23:08:34.554329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove non-ascii characters from the text\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: remove_non_ascii(x))\n\n# double check\nprint(train_df[\"text\"][38])\nprint(train_df[\"text_clean\"][38])\nprint(train_df[\"text\"][7586])\nprint(train_df[\"text_clean\"][7586])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.556415Z","iopub.execute_input":"2026-01-16T23:08:34.556653Z","iopub.status.idle":"2026-01-16T23:08:34.580833Z","shell.execute_reply.started":"2026-01-16T23:08:34.556629Z","shell.execute_reply":"2026-01-16T23:08:34.580043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Remove_special_characters\"></a>\n\n### Remove special characters: \nThe special characters could be symbols, emojis, and other graphic characters.\nWe use the \"Toxic Comment Classification Challenge\" dataset as the \"Real or Not? NLP with Disaster Tweets\" dataset do not have any special charaters in their text.\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"train_df_jtcc = pd.read_csv(\"/kaggle/input/jigsaw-toxic-comment-classification-challenge/train.csv.zip\")\nprint(train_df_jtcc.shape)\ntrain_df_jtcc.head()","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:34.582394Z","iopub.execute_input":"2026-01-16T23:08:34.582760Z","iopub.status.idle":"2026-01-16T23:08:35.729369Z","shell.execute_reply.started":"2026-01-16T23:08:34.582725Z","shell.execute_reply":"2026-01-16T23:08:35.728633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_special_characters(text):\n    \"\"\"\n        Remove special special characters, including symbols, emojis, and other graphic characters\n    \"\"\"\n    emoji_pattern = re.compile(\n        '['\n        u'\\U0001F600-\\U0001F64F'  # emoticons\n        u'\\U0001F300-\\U0001F5FF'  # symbols & pictographs\n        u'\\U0001F680-\\U0001F6FF'  # transport & map symbols\n        u'\\U0001F1E0-\\U0001F1FF'  # flags (iOS)\n        u'\\U00002702-\\U000027B0'\n        u'\\U000024C2-\\U0001F251'\n        ']+',\n        flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:35.730702Z","iopub.execute_input":"2026-01-16T23:08:35.731014Z","iopub.status.idle":"2026-01-16T23:08:35.735670Z","shell.execute_reply.started":"2026-01-16T23:08:35.730986Z","shell.execute_reply":"2026-01-16T23:08:35.734869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\n# remove non-ascii characters from the text\ntrain_df_jtcc[\"text_clean\"] = train_df_jtcc[\"comment_text\"].apply(lambda x: remove_special_characters(x))\ndisplay(train_df_jtcc.head())\n\n# double check\nprint(train_df_jtcc[\"comment_text\"][143])\nprint(train_df_jtcc[\"text_clean\"][143])\nprint(train_df_jtcc[\"comment_text\"][189])\nprint(train_df_jtcc[\"text_clean\"][189])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:35.737113Z","iopub.execute_input":"2026-01-16T23:08:35.737366Z","iopub.status.idle":"2026-01-16T23:08:37.699159Z","shell.execute_reply.started":"2026-01-16T23:08:35.737337Z","shell.execute_reply":"2026-01-16T23:08:37.698459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Saving disk space\ndel train_df_jtcc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:37.700496Z","iopub.execute_input":"2026-01-16T23:08:37.700748Z","iopub.status.idle":"2026-01-16T23:08:37.704952Z","shell.execute_reply.started":"2026-01-16T23:08:37.700722Z","shell.execute_reply":"2026-01-16T23:08:37.704163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Remove_punctuations\"></a>\n\n## Remove punctuations:\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"def remove_punct(text):\n    \"\"\"\n        Remove the punctuation\n    \"\"\"\n#     return re.sub(r'[]!\"$%&\\'()*+,./:;=#@?[\\\\^_`{|}~-]+', \"\", text)\n    return text.translate(str.maketrans('', '', string.punctuation))","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:37.706210Z","iopub.execute_input":"2026-01-16T23:08:37.706575Z","iopub.status.idle":"2026-01-16T23:08:37.718659Z","shell.execute_reply.started":"2026-01-16T23:08:37.706538Z","shell.execute_reply":"2026-01-16T23:08:37.718084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove punctuations from the text\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: remove_punct(x))\n\n# double check\nprint(train_df[\"text\"][5])\nprint(train_df[\"text_clean\"][5])\nprint(train_df[\"text\"][7597])\nprint(train_df[\"text_clean\"][7597])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:37.719908Z","iopub.execute_input":"2026-01-16T23:08:37.720126Z","iopub.status.idle":"2026-01-16T23:08:37.765947Z","shell.execute_reply.started":"2026-01-16T23:08:37.720105Z","shell.execute_reply":"2026-01-16T23:08:37.765261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Other_Manual_Text_Cleaning_Tasks\"></a>\n\n## Other Manual Text Cleaning Tasks: \n\nOther techniques could be considered and manually processed case by case: \n    - Replace the Unicode character with equivalent ASCII character (instead of removing)\n    - Replace the entity references with their actual symbols  instead of removing as HTML tags\n    - Replace the Typos, slang, acronyms or informal abbreviations - depend on different situations or main topics of the NLP such as finance or medical topics.\n    - List out all the hashtags/ usernames then replace with equivalent words\n    - Replace the emoticon/ emoji with equivalant word meaning such as \":)\" with \"smile\" \n    - Spelling correction\n\n<a id=\"Replace_Typos\"></a>\n### Replace the Typos, slang, acronyms or informal abbreviations: \n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"def other_clean(text):\n        \"\"\"\n            Other manual text cleaning techniques\n        \"\"\"\n        # Typos, slang and other\n        sample_typos_slang = {\n                                \"w/e\": \"whatever\",\n                                \"usagov\": \"usa government\",\n                                \"recentlu\": \"recently\",\n                                \"ph0tos\": \"photos\",\n                                \"amirite\": \"am i right\",\n                                \"exp0sed\": \"exposed\",\n                                \"<3\": \"love\",\n                                \"luv\": \"love\",\n                                \"amageddon\": \"armageddon\",\n                                \"trfc\": \"traffic\",\n                                \"16yr\": \"16 year\"\n                                }\n\n        # Acronyms\n        sample_acronyms =  { \n                            \"mh370\": \"malaysia airlines flight 370\",\n                            \"okwx\": \"oklahoma city weather\",\n                            \"arwx\": \"arkansas weather\",    \n                            \"gawx\": \"georgia weather\",  \n                            \"scwx\": \"south carolina weather\",  \n                            \"cawx\": \"california weather\",\n                            \"tnwx\": \"tennessee weather\",\n                            \"azwx\": \"arizona weather\",  \n                            \"alwx\": \"alabama weather\",\n                            \"usnwsgov\": \"united states national weather service\",\n                            \"2mw\": \"tomorrow\"\n                            }\n\n        \n        # Some common abbreviations \n        sample_abbr = {\n                        \"$\" : \" dollar \",\n                        \"€\" : \" euro \",\n                        \"4ao\" : \"for adults only\",\n                        \"a.m\" : \"before midday\",\n                        \"a3\" : \"anytime anywhere anyplace\",\n                        \"aamof\" : \"as a matter of fact\",\n                        \"acct\" : \"account\",\n                        \"adih\" : \"another day in hell\",\n                        \"afaic\" : \"as far as i am concerned\",\n                        \"afaict\" : \"as far as i can tell\",\n                        \"afaik\" : \"as far as i know\",\n                        \"afair\" : \"as far as i remember\",\n                        \"afk\" : \"away from keyboard\",\n                        \"app\" : \"application\",\n                        \"approx\" : \"approximately\",\n                        \"apps\" : \"applications\",\n                        \"asap\" : \"as soon as possible\",\n                        \"asl\" : \"age, sex, location\",\n                        \"atk\" : \"at the keyboard\",\n                        \"ave.\" : \"avenue\",\n                        \"aymm\" : \"are you my mother\",\n                        \"ayor\" : \"at your own risk\", \n                        \"b&b\" : \"bed and breakfast\",\n                        \"b+b\" : \"bed and breakfast\",\n                        \"b.c\" : \"before christ\",\n                        \"b2b\" : \"business to business\",\n                        \"b2c\" : \"business to customer\",\n                        \"b4\" : \"before\",\n                        \"b4n\" : \"bye for now\",\n                        \"b@u\" : \"back at you\",\n                        \"bae\" : \"before anyone else\",\n                        \"bak\" : \"back at keyboard\",\n                        \"bbbg\" : \"bye bye be good\",\n                        \"bbc\" : \"british broadcasting corporation\",\n                        \"bbias\" : \"be back in a second\",\n                        \"bbl\" : \"be back later\",\n                        \"bbs\" : \"be back soon\",\n                        \"be4\" : \"before\",\n                        \"bfn\" : \"bye for now\",\n                        \"blvd\" : \"boulevard\",\n                        \"bout\" : \"about\",\n                        \"brb\" : \"be right back\",\n                        \"bros\" : \"brothers\",\n                        \"brt\" : \"be right there\",\n                        \"bsaaw\" : \"big smile and a wink\",\n                        \"btw\" : \"by the way\",\n                        \"bwl\" : \"bursting with laughter\",\n                        \"c/o\" : \"care of\",\n                        \"cet\" : \"central european time\",\n                        \"cf\" : \"compare\",\n                        \"cia\" : \"central intelligence agency\",\n                        \"csl\" : \"can not stop laughing\",\n                        \"cu\" : \"see you\",\n                        \"cul8r\" : \"see you later\",\n                        \"cv\" : \"curriculum vitae\",\n                        \"cwot\" : \"complete waste of time\",\n                        \"cya\" : \"see you\",\n                        \"cyt\" : \"see you tomorrow\",\n                        \"dae\" : \"does anyone else\",\n                        \"dbmib\" : \"do not bother me i am busy\",\n                        \"diy\" : \"do it yourself\",\n                        \"dm\" : \"direct message\",\n                        \"dwh\" : \"during work hours\",\n                        \"e123\" : \"easy as one two three\",\n                        \"eet\" : \"eastern european time\",\n                        \"eg\" : \"example\",\n                        \"embm\" : \"early morning business meeting\",\n                        \"encl\" : \"enclosed\",\n                        \"encl.\" : \"enclosed\",\n                        \"etc\" : \"and so on\",\n                        \"faq\" : \"frequently asked questions\",\n                        \"fawc\" : \"for anyone who cares\",\n                        \"fb\" : \"facebook\",\n                        \"fc\" : \"fingers crossed\",\n                        \"fig\" : \"figure\",\n                        \"fimh\" : \"forever in my heart\", \n                        \"ft.\" : \"feet\",\n                        \"ft\" : \"featuring\",\n                        \"ftl\" : \"for the loss\",\n                        \"ftw\" : \"for the win\",\n                        \"fwiw\" : \"for what it is worth\",\n                        \"fyi\" : \"for your information\",\n                        \"g9\" : \"genius\",\n                        \"gahoy\" : \"get a hold of yourself\",\n                        \"gal\" : \"get a life\",\n                        \"gcse\" : \"general certificate of secondary education\",\n                        \"gfn\" : \"gone for now\",\n                        \"gg\" : \"good game\",\n                        \"gl\" : \"good luck\",\n                        \"glhf\" : \"good luck have fun\",\n                        \"gmt\" : \"greenwich mean time\",\n                        \"gmta\" : \"great minds think alike\",\n                        \"gn\" : \"good night\",\n                        \"g.o.a.t\" : \"greatest of all time\",\n                        \"goat\" : \"greatest of all time\",\n                        \"goi\" : \"get over it\",\n                        \"gps\" : \"global positioning system\",\n                        \"gr8\" : \"great\",\n                        \"gratz\" : \"congratulations\",\n                        \"gyal\" : \"girl\",\n                        \"h&c\" : \"hot and cold\",\n                        \"hp\" : \"horsepower\",\n                        \"hr\" : \"hour\",\n                        \"hrh\" : \"his royal highness\",\n                        \"ht\" : \"height\",\n                        \"ibrb\" : \"i will be right back\",\n                        \"ic\" : \"i see\",\n                        \"icq\" : \"i seek you\",\n                        \"icymi\" : \"in case you missed it\",\n                        \"idc\" : \"i do not care\",\n                        \"idgadf\" : \"i do not give a damn fuck\",\n                        \"idgaf\" : \"i do not give a fuck\",\n                        \"idk\" : \"i do not know\",\n                        \"ie\" : \"that is\",\n                        \"i.e\" : \"that is\",\n                        \"ifyp\" : \"i feel your pain\",\n                        \"IG\" : \"instagram\",\n                        \"iirc\" : \"if i remember correctly\",\n                        \"ilu\" : \"i love you\",\n                        \"ily\" : \"i love you\",\n                        \"imho\" : \"in my humble opinion\",\n                        \"imo\" : \"in my opinion\",\n                        \"imu\" : \"i miss you\",\n                        \"iow\" : \"in other words\",\n                        \"irl\" : \"in real life\",\n                        \"j4f\" : \"just for fun\",\n                        \"jic\" : \"just in case\",\n                        \"jk\" : \"just kidding\",\n                        \"jsyk\" : \"just so you know\",\n                        \"l8r\" : \"later\",\n                        \"lb\" : \"pound\",\n                        \"lbs\" : \"pounds\",\n                        \"ldr\" : \"long distance relationship\",\n                        \"lmao\" : \"laugh my ass off\",\n                        \"lmfao\" : \"laugh my fucking ass off\",\n                        \"lol\" : \"laughing out loud\",\n                        \"ltd\" : \"limited\",\n                        \"ltns\" : \"long time no see\",\n                        \"m8\" : \"mate\",\n                        \"mf\" : \"motherfucker\",\n                        \"mfs\" : \"motherfuckers\",\n                        \"mfw\" : \"my face when\",\n                        \"mofo\" : \"motherfucker\",\n                        \"mph\" : \"miles per hour\",\n                        \"mr\" : \"mister\",\n                        \"mrw\" : \"my reaction when\",\n                        \"ms\" : \"miss\",\n                        \"mte\" : \"my thoughts exactly\",\n                        \"nagi\" : \"not a good idea\",\n                        \"nbc\" : \"national broadcasting company\",\n                        \"nbd\" : \"not big deal\",\n                        \"nfs\" : \"not for sale\",\n                        \"ngl\" : \"not going to lie\",\n                        \"nhs\" : \"national health service\",\n                        \"nrn\" : \"no reply necessary\",\n                        \"nsfl\" : \"not safe for life\",\n                        \"nsfw\" : \"not safe for work\",\n                        \"nth\" : \"nice to have\",\n                        \"nvr\" : \"never\",\n                        \"nyc\" : \"new york city\",\n                        \"oc\" : \"original content\",\n                        \"og\" : \"original\",\n                        \"ohp\" : \"overhead projector\",\n                        \"oic\" : \"oh i see\",\n                        \"omdb\" : \"over my dead body\",\n                        \"omg\" : \"oh my god\",\n                        \"omw\" : \"on my way\",\n                        \"p.a\" : \"per annum\",\n                        \"p.m\" : \"after midday\",\n                        \"pm\" : \"prime minister\",\n                        \"poc\" : \"people of color\",\n                        \"pov\" : \"point of view\",\n                        \"pp\" : \"pages\",\n                        \"ppl\" : \"people\",\n                        \"prw\" : \"parents are watching\",\n                        \"ps\" : \"postscript\",\n                        \"pt\" : \"point\",\n                        \"ptb\" : \"please text back\",\n                        \"pto\" : \"please turn over\",\n                        \"qpsa\" : \"what happens\", #\"que pasa\",\n                        \"ratchet\" : \"rude\",\n                        \"rbtl\" : \"read between the lines\",\n                        \"rlrt\" : \"real life retweet\", \n                        \"rofl\" : \"rolling on the floor laughing\",\n                        \"roflol\" : \"rolling on the floor laughing out loud\",\n                        \"rotflmao\" : \"rolling on the floor laughing my ass off\",\n                        \"rt\" : \"retweet\",\n                        \"ruok\" : \"are you ok\",\n                        \"sfw\" : \"safe for work\",\n                        \"sk8\" : \"skate\",\n                        \"smh\" : \"shake my head\",\n                        \"sq\" : \"square\",\n                        \"srsly\" : \"seriously\", \n                        \"ssdd\" : \"same stuff different day\",\n                        \"tbh\" : \"to be honest\",\n                        \"tbs\" : \"tablespooful\",\n                        \"tbsp\" : \"tablespooful\",\n                        \"tfw\" : \"that feeling when\",\n                        \"thks\" : \"thank you\",\n                        \"tho\" : \"though\",\n                        \"thx\" : \"thank you\",\n                        \"tia\" : \"thanks in advance\",\n                        \"til\" : \"today i learned\",\n                        \"tl;dr\" : \"too long i did not read\",\n                        \"tldr\" : \"too long i did not read\",\n                        \"tmb\" : \"tweet me back\",\n                        \"tntl\" : \"trying not to laugh\",\n                        \"ttyl\" : \"talk to you later\",\n                        \"u\" : \"you\",\n                        \"u2\" : \"you too\",\n                        \"u4e\" : \"yours for ever\",\n                        \"utc\" : \"coordinated universal time\",\n                        \"w/\" : \"with\",\n                        \"w/o\" : \"without\",\n                        \"w8\" : \"wait\",\n                        \"wassup\" : \"what is up\",\n                        \"wb\" : \"welcome back\",\n                        \"wtf\" : \"what the fuck\",\n                        \"wtg\" : \"way to go\",\n                        \"wtpa\" : \"where the party at\",\n                        \"wuf\" : \"where are you from\",\n                        \"wuzup\" : \"what is up\",\n                        \"wywh\" : \"wish you were here\",\n                        \"yd\" : \"yard\",\n                        \"ygtr\" : \"you got that right\",\n                        \"ynk\" : \"you never know\",\n                        \"zzz\" : \"sleeping bored and tired\"\n                        }\n            \n        sample_typos_slang_pattern = re.compile(r'(?<!\\w)(' + '|'.join(re.escape(key) for key in sample_typos_slang.keys()) + r')(?!\\w)')\n        sample_acronyms_pattern = re.compile(r'(?<!\\w)(' + '|'.join(re.escape(key) for key in sample_acronyms.keys()) + r')(?!\\w)')\n        sample_abbr_pattern = re.compile(r'(?<!\\w)(' + '|'.join(re.escape(key) for key in sample_abbr.keys()) + r')(?!\\w)')\n        \n        text = sample_typos_slang_pattern.sub(lambda x: sample_typos_slang[x.group()], text)\n        text = sample_acronyms_pattern.sub(lambda x: sample_acronyms[x.group()], text)\n        text = sample_abbr_pattern.sub(lambda x: sample_abbr[x.group()], text)\n        \n        return text","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:37.768976Z","iopub.execute_input":"2026-01-16T23:08:37.769213Z","iopub.status.idle":"2026-01-16T23:08:37.800250Z","shell.execute_reply.started":"2026-01-16T23:08:37.769191Z","shell.execute_reply":"2026-01-16T23:08:37.799355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\n\n# Test\ntest_text = \"\"\"\n            brb with some sample ph0tos I lov u. I need some $ for 2mw.\n            \"\"\"\nprint(\"Test: \", other_clean(test_text))\n\n# remove punctuations from the text\ntrain_df[\"text_clean\"] = train_df[\"text_clean\"].apply(lambda x: other_clean(x))\n\n# double check\nprint(train_df[\"text\"][1844])\nprint(train_df[\"text_clean\"][1844])\nprint(train_df[\"text\"][4409])\nprint(train_df[\"text_clean\"][4409])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:37.801524Z","iopub.execute_input":"2026-01-16T23:08:37.801874Z","iopub.status.idle":"2026-01-16T23:08:39.221754Z","shell.execute_reply.started":"2026-01-16T23:08:37.801839Z","shell.execute_reply":"2026-01-16T23:08:39.220887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Spelling_correction\"></a>\n\n### Spelling Correction\nSpelling correction could also be considered an optional preprocessing task as the social media text data is often are typos or mistyped. However, the spelling correction output should be carefully double-checked with the original text input as it could be a mistake.\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"from textblob import TextBlob\nprint(\"Test: \", TextBlob(\"sleapy and tehre is no plaxe I'm gioong to.\").correct())","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:39.223191Z","iopub.execute_input":"2026-01-16T23:08:39.223469Z","iopub.status.idle":"2026-01-16T23:08:39.354623Z","shell.execute_reply.started":"2026-01-16T23:08:39.223443Z","shell.execute_reply":"2026-01-16T23:08:39.353943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Text_Preprocessing\"></a>\n\n# Text Preprocessing:\n\n<a id=\"Tokenization\"></a>\n## Tokenization\nTokenization is a common technique that split a sentence into tokens, where a token could be characters, words, phrases, symbols, or other meaningful elements. By breaking sentences into smaller chunks, that would help to investigate the words in a sentence and also the subsequent steps in the NLP pipeline, such as stemming. \n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"# Tokenizing the tweet base texts.\nfrom nltk.tokenize import word_tokenize\n\ntrain_df['tokenized'] = train_df['text_clean'].apply(word_tokenize)\ntrain_df.head()","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:39.356014Z","iopub.execute_input":"2026-01-16T23:08:39.356263Z","iopub.status.idle":"2026-01-16T23:08:40.223704Z","shell.execute_reply.started":"2026-01-16T23:08:39.356236Z","shell.execute_reply":"2026-01-16T23:08:40.222981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Removing stopwords.\nnltk.download(\"stopwords\")\nfrom nltk.corpus import stopwords\n\nstop = set(stopwords.words('english'))\ntrain_df['stopwords_removed'] = train_df['tokenized'].apply(lambda x: [word for word in x if word not in stop])\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:40.225167Z","iopub.execute_input":"2026-01-16T23:08:40.225516Z","iopub.status.idle":"2026-01-16T23:08:40.267374Z","shell.execute_reply.started":"2026-01-16T23:08:40.225478Z","shell.execute_reply":"2026-01-16T23:08:40.266589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.stem import PorterStemmer\n\ndef porter_stemmer(text):\n    \"\"\"\n        Stem words in list of tokenized words with PorterStemmer\n    \"\"\"\n    stemmer = nltk.PorterStemmer()\n    stems = [stemmer.stem(i) for i in text]\n    return stems","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:40.268622Z","iopub.execute_input":"2026-01-16T23:08:40.268888Z","iopub.status.idle":"2026-01-16T23:08:40.273102Z","shell.execute_reply.started":"2026-01-16T23:08:40.268861Z","shell.execute_reply":"2026-01-16T23:08:40.272428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\ntrain_df['porter_stemmer'] = train_df['stopwords_removed'].apply(lambda x: porter_stemmer(x))\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:40.274200Z","iopub.execute_input":"2026-01-16T23:08:40.274447Z","iopub.status.idle":"2026-01-16T23:08:41.654768Z","shell.execute_reply.started":"2026-01-16T23:08:40.274424Z","shell.execute_reply":"2026-01-16T23:08:41.654060Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"SnowballStemmer\"></a>\n### SnowballStemmer\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"from nltk.stem import SnowballStemmer\n\ndef snowball_stemmer(text):\n    \"\"\"\n        Stem words in list of tokenized words with SnowballStemmer\n    \"\"\"\n    stemmer = nltk.SnowballStemmer(\"english\")\n    stems = [stemmer.stem(i) for i in text]\n    return stems","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:41.656133Z","iopub.execute_input":"2026-01-16T23:08:41.656358Z","iopub.status.idle":"2026-01-16T23:08:41.660680Z","shell.execute_reply.started":"2026-01-16T23:08:41.656337Z","shell.execute_reply":"2026-01-16T23:08:41.659860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\ntrain_df['snowball_stemmer'] = train_df['stopwords_removed'].apply(lambda x: snowball_stemmer(x))\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:41.661910Z","iopub.execute_input":"2026-01-16T23:08:41.662130Z","iopub.status.idle":"2026-01-16T23:08:42.617382Z","shell.execute_reply.started":"2026-01-16T23:08:41.662108Z","shell.execute_reply":"2026-01-16T23:08:42.616566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"LancasterStemmer\"></a>\n### LancasterStemmer\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"from nltk.stem import LancasterStemmer\n\ndef lancaster_stemmer(text):\n    \"\"\"\n        Stem words in list of tokenized words with LancasterStemmer\n    \"\"\"\n    stemmer = nltk.LancasterStemmer()\n    stems = [stemmer.stem(i) for i in text]\n    return stems","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:42.618758Z","iopub.execute_input":"2026-01-16T23:08:42.619058Z","iopub.status.idle":"2026-01-16T23:08:42.623385Z","shell.execute_reply.started":"2026-01-16T23:08:42.619032Z","shell.execute_reply":"2026-01-16T23:08:42.622561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\ntrain_df['lancaster_stemmer'] = train_df['stopwords_removed'].apply(lambda x: lancaster_stemmer(x))\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:42.624685Z","iopub.execute_input":"2026-01-16T23:08:42.625030Z","iopub.status.idle":"2026-01-16T23:08:44.172157Z","shell.execute_reply.started":"2026-01-16T23:08:42.625001Z","shell.execute_reply":"2026-01-16T23:08:44.171320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.corpus import wordnet\nfrom nltk.corpus import brown\n\nwordnet_map = {\"N\":wordnet.NOUN, \n               \"V\":wordnet.VERB, \n               \"J\":wordnet.ADJ, \n               \"R\":wordnet.ADV\n              }\n    \ntrain_sents = brown.tagged_sents(categories='news')\nt0 = nltk.DefaultTagger('NN')\nt1 = nltk.UnigramTagger(train_sents, backoff=t0)\nt2 = nltk.BigramTagger(train_sents, backoff=t1)\n\ndef pos_tag_wordnet(text, pos_tag_type=\"pos_tag\"):\n    \"\"\"\n        Create pos_tag with wordnet format\n    \"\"\"\n    pos_tagged_text = t2.tag(text)\n    \n    # map the pos tagging output with wordnet output \n    pos_tagged_text = [(word, wordnet_map.get(pos_tag[0])) if pos_tag[0] in wordnet_map.keys() else (word, wordnet.NOUN) for (word, pos_tag) in pos_tagged_text ]\n    return pos_tagged_text","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:44.173372Z","iopub.execute_input":"2026-01-16T23:08:44.173594Z","iopub.status.idle":"2026-01-16T23:08:45.779512Z","shell.execute_reply.started":"2026-01-16T23:08:44.173573Z","shell.execute_reply":"2026-01-16T23:08:45.778825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pos_tag_wordnet(train_df['stopwords_removed'][2])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:45.780529Z","iopub.execute_input":"2026-01-16T23:08:45.780749Z","iopub.status.idle":"2026-01-16T23:08:45.786352Z","shell.execute_reply.started":"2026-01-16T23:08:45.780726Z","shell.execute_reply":"2026-01-16T23:08:45.785600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\ntrain_df['combined_postag_wnet'] = train_df['stopwords_removed'].apply(lambda x: pos_tag_wordnet(x))\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:45.792042Z","iopub.execute_input":"2026-01-16T23:08:45.792483Z","iopub.status.idle":"2026-01-16T23:08:46.014703Z","shell.execute_reply.started":"2026-01-16T23:08:45.792449Z","shell.execute_reply":"2026-01-16T23:08:46.014003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\n\ndef lemmatize_word(text):\n    \"\"\"\n        Lemmatize the tokenized words\n    \"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    lemma = [lemmatizer.lemmatize(word, tag) for word, tag in text]\n    return lemma","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.016271Z","iopub.execute_input":"2026-01-16T23:08:46.016512Z","iopub.status.idle":"2026-01-16T23:08:46.020554Z","shell.execute_reply.started":"2026-01-16T23:08:46.016486Z","shell.execute_reply":"2026-01-16T23:08:46.019930Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Lemmatization_wo_pos\"></a>\n\n### Lemmatization without POS Tagging:\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"%time \n\n# Test without POS Tagging\nlemmatizer = WordNetLemmatizer()\n\ntrain_df['lemmatize_word_wo_pos'] = train_df['stopwords_removed'].apply(lambda x: [lemmatizer.lemmatize(word) for word in x])\ntrain_df['lemmatize_word_wo_pos'] = train_df['lemmatize_word_wo_pos'].apply(lambda x: [word for word in x if word not in stop])\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.021652Z","iopub.execute_input":"2026-01-16T23:08:46.021894Z","iopub.status.idle":"2026-01-16T23:08:46.320153Z","shell.execute_reply.started":"2026-01-16T23:08:46.021870Z","shell.execute_reply":"2026-01-16T23:08:46.319311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df[\"combined_postag_wnet\"][8])\nprint(train_df[\"lemmatize_word_wo_pos\"][8])","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.321471Z","iopub.execute_input":"2026-01-16T23:08:46.321701Z","iopub.status.idle":"2026-01-16T23:08:46.326151Z","shell.execute_reply.started":"2026-01-16T23:08:46.321680Z","shell.execute_reply":"2026-01-16T23:08:46.325351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Lemmatization_w_pos\"></a>\n\n### Lemmatization with POS Tagging:\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"%time \n\n# Test with POS Tagging\nlemmatizer = WordNetLemmatizer()\n\ntrain_df['lemmatize_word_w_pos'] = train_df['combined_postag_wnet'].apply(lambda x: lemmatize_word(x))\ntrain_df['lemmatize_word_w_pos'] = train_df['lemmatize_word_w_pos'].apply(lambda x: [word for word in x if word not in stop]) # double check to remove stop words\ntrain_df['lemmatize_text'] = [' '.join(map(str, l)) for l in train_df['lemmatize_word_w_pos']] # join back to text\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.327254Z","iopub.execute_input":"2026-01-16T23:08:46.327510Z","iopub.status.idle":"2026-01-16T23:08:46.655828Z","shell.execute_reply.started":"2026-01-16T23:08:46.327486Z","shell.execute_reply":"2026-01-16T23:08:46.655093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Comparing the output of Lemmatization on non-POS-Tagging and POS-Tagging output. We can see in the original text, the word \\\" happening\\\" is a verb and was corrected assigned as a verb by POS-tagging stage, then Lemmatize accurately with back as \\\"happen\\\" but lemmatized without-POS-tagging resulted in \\\"happening\\\" is not correct. ","metadata":{}},{"cell_type":"code","source":"print(train_df[\"text\"][8])\nprint(train_df[\"combined_postag_wnet\"][8])\nprint(train_df[\"lemmatize_word_wo_pos\"][8])\nprint(train_df[\"lemmatize_word_w_pos\"][8])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.656953Z","iopub.execute_input":"2026-01-16T23:08:46.657197Z","iopub.status.idle":"2026-01-16T23:08:46.662573Z","shell.execute_reply.started":"2026-01-16T23:08:46.657173Z","shell.execute_reply":"2026-01-16T23:08:46.661826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Comparison between original text and the lammatized text:","metadata":{}},{"cell_type":"code","source":"display(train_df[\"text\"][0], train_df[\"lemmatize_text\"][0])\ndisplay(train_df[\"text\"][5], train_df[\"lemmatize_text\"][5])\ndisplay(train_df[\"text\"][10], train_df[\"lemmatize_text\"][10])\ndisplay(train_df[\"text\"][15], train_df[\"lemmatize_text\"][15])\ndisplay(train_df[\"text\"][20], train_df[\"lemmatize_text\"][20])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.663976Z","iopub.execute_input":"2026-01-16T23:08:46.664373Z","iopub.status.idle":"2026-01-16T23:08:46.688531Z","shell.execute_reply.started":"2026-01-16T23:08:46.664320Z","shell.execute_reply":"2026-01-16T23:08:46.687736Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id=\"Other_Text_Preprocessing\"></a>\n\n## Other (Optional) Text Preprocessing Techniques:\n- language detection\n- Code mixing and transliteration\n\n<a id=\"Language_Detection\"></a>\n### Language Detection:\nWe will use the package [polyglot](https://github.com/aboSamoor/polyglot) for language detection\n\n[Back To Table of Contents](#top_section)","metadata":{}},{"cell_type":"code","source":"# Install the main polygot and other neccesary packages\n!pip install pyicu\n!pip install pycld2\n!pip install polyglot","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:08:46.689592Z","iopub.execute_input":"2026-01-16T23:08:46.689846Z","iopub.status.idle":"2026-01-16T23:09:01.417920Z","shell.execute_reply.started":"2026-01-16T23:08:46.689818Z","shell.execute_reply":"2026-01-16T23:09:01.417128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We will use the \"Jigsaw Multilingual Toxic Comment Classification\" dataset for this case as the dataset is multilingual","metadata":{}},{"cell_type":"code","source":"train_df_jmtc = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nprint(train_df_jmtc.shape)\ntrain_df_jmtc.head()","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:09:01.419391Z","iopub.execute_input":"2026-01-16T23:09:01.419813Z","iopub.status.idle":"2026-01-16T23:09:02.418568Z","shell.execute_reply.started":"2026-01-16T23:09:01.419743Z","shell.execute_reply":"2026-01-16T23:09:02.417837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install transformers fasttext\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:12:39.198414Z","iopub.execute_input":"2026-01-16T23:12:39.198767Z","iopub.status.idle":"2026-01-16T23:12:46.287686Z","shell.execute_reply.started":"2026-01-16T23:12:39.198730Z","shell.execute_reply":"2026-01-16T23:12:46.286697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import fasttext\nimport urllib.request\nimport os\n\nif not os.path.exists(\"lid.176.bin\"):\n    url = \"https://dl.fbaipublicfiles.com/fasttext/supervised-models/lid.176.bin\"\n    urllib.request.urlretrieve(url, \"lid.176.bin\")\n\nlang_model = fasttext.load_model(\"lid.176.bin\")\n\ndef get_language(text):\n    text = \"\".join(x for x in str(text) if x.isprintable())\n    label, prob = lang_model.predict(text)\n    return label[0].replace(\"__label__\", \"\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:12:58.250039Z","iopub.execute_input":"2026-01-16T23:12:58.250381Z","iopub.status.idle":"2026-01-16T23:12:59.107946Z","shell.execute_reply.started":"2026-01-16T23:12:58.250341Z","shell.execute_reply":"2026-01-16T23:12:59.107176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_jmtc[\"lang\"] = train_df_jmtc[\"comment_text\"].apply(get_language)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:13.777409Z","iopub.execute_input":"2026-01-16T23:13:13.777707Z","iopub.status.idle":"2026-01-16T23:13:33.463416Z","shell.execute_reply.started":"2026-01-16T23:13:13.777682Z","shell.execute_reply":"2026-01-16T23:13:33.462670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_jmtc = train_df_jmtc[train_df_jmtc[\"lang\"] == \"en\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:37.225618Z","iopub.execute_input":"2026-01-16T23:13:37.225941Z","iopub.status.idle":"2026-01-16T23:13:37.308114Z","shell.execute_reply.started":"2026-01-16T23:13:37.225913Z","shell.execute_reply":"2026-01-16T23:13:37.307379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save disk space\ndel train_df_jmtc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:40.108309Z","iopub.execute_input":"2026-01-16T23:13:40.108595Z","iopub.status.idle":"2026-01-16T23:13:40.121256Z","shell.execute_reply.started":"2026-01-16T23:13:40.108571Z","shell.execute_reply":"2026-01-16T23:13:40.120456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\ndef cv(data, ngram = 1, MAX_NB_WORDS = 75000):\n    count_vectorizer = CountVectorizer(ngram_range = (ngram, ngram), max_features = MAX_NB_WORDS)\n    emb = count_vectorizer.fit_transform(data).toarray()\n    print(\"count vectorize with\", str(np.array(emb).shape[1]), \"features\")\n    return emb, count_vectorizer","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:44.610535Z","iopub.execute_input":"2026-01-16T23:13:44.610868Z","iopub.status.idle":"2026-01-16T23:13:44.616387Z","shell.execute_reply.started":"2026-01-16T23:13:44.610831Z","shell.execute_reply":"2026-01-16T23:13:44.615603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_out(emb, feat, ngram, compared_sentence=0):\n    print(ngram,\"bag-of-words: \")\n    print(feat.get_feature_names(), \"\\n\")\n    print(ngram,\"bag-of-feature: \")\n    print(test_cv_1gram.vocabulary_, \"\\n\")\n    print(\"BoW matrix:\")\n    print(pd.DataFrame(emb.transpose(), index = feat.get_feature_names()).head(), \"\\n\")\n    print(ngram,\"vector example:\")\n    print(train_df[\"lemmatize_text\"][compared_sentence])\n    print(emb[compared_sentence], \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:47.318718Z","iopub.execute_input":"2026-01-16T23:13:47.319065Z","iopub.status.idle":"2026-01-16T23:13:47.324713Z","shell.execute_reply.started":"2026-01-16T23:13:47.319035Z","shell.execute_reply":"2026-01-16T23:13:47.323983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_corpus = train_df[\"lemmatize_text\"][:5].tolist()\nprint(\"The test corpus: \", test_corpus, \"\\n\")\n\ntest_cv_em_1gram, test_cv_1gram = cv(test_corpus, ngram=1)\nprint_out(test_cv_em_1gram, test_cv_1gram, ngram=\"Uni-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:50.908692Z","iopub.execute_input":"2026-01-16T23:13:50.909033Z","iopub.status.idle":"2026-01-16T23:13:50.920654Z","shell.execute_reply.started":"2026-01-16T23:13:50.908994Z","shell.execute_reply":"2026-01-16T23:13:50.919919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_cv_em_2gram, test_cv_2gram = cv(test_corpus, ngram=2)\nprint_out(test_cv_em_2gram, test_cv_2gram, ngram=\"Bi-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:54.442602Z","iopub.execute_input":"2026-01-16T23:13:54.442906Z","iopub.status.idle":"2026-01-16T23:13:54.452711Z","shell.execute_reply.started":"2026-01-16T23:13:54.442878Z","shell.execute_reply":"2026-01-16T23:13:54.451938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_cv_em_3gram, test_cv_3gram = cv(test_corpus, ngram=3)\nprint_out(test_cv_em_2gram, test_cv_2gram, ngram=\"Tri-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:13:57.137371Z","iopub.execute_input":"2026-01-16T23:13:57.137674Z","iopub.status.idle":"2026-01-16T23:13:57.147894Z","shell.execute_reply.started":"2026-01-16T23:13:57.137646Z","shell.execute_reply":"2026-01-16T23:13:57.147015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\n# implement into the whole dataset\ntrain_df_corpus = train_df[\"lemmatize_text\"].tolist()\ntrain_df_em_1gram, vc_1gram = cv(train_df_corpus, 1)\ntrain_df_em_2gram, vc_2gram = cv(train_df_corpus, 2)\ntrain_df_em_3gram, vc_3gram = cv(train_df_corpus, 3)\n\nprint(len(train_df_corpus))\nprint(train_df_em_1gram.shape)\nprint(train_df_em_2gram.shape)\nprint(train_df_em_3gram.shape)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:00.067611Z","iopub.execute_input":"2026-01-16T23:14:00.067987Z","iopub.status.idle":"2026-01-16T23:14:04.970185Z","shell.execute_reply.started":"2026-01-16T23:14:00.067956Z","shell.execute_reply":"2026-01-16T23:14:04.969287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_df_em_1gram, train_df_em_2gram, train_df_em_3gram","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:10.015676Z","iopub.execute_input":"2026-01-16T23:14:10.016039Z","iopub.status.idle":"2026-01-16T23:14:10.096394Z","shell.execute_reply.started":"2026-01-16T23:14:10.016005Z","shell.execute_reply":"2026-01-16T23:14:10.095462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ndef TFIDF(data, ngram = 1, MAX_NB_WORDS = 75000):\n    tfidf_x = TfidfVectorizer(ngram_range = (ngram, ngram), max_features = MAX_NB_WORDS)\n    emb = tfidf_x.fit_transform(data).toarray()\n    print(\"tf-idf with\", str(np.array(emb).shape[1]), \"features\")\n    return emb, tfidf_x","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:11.995932Z","iopub.execute_input":"2026-01-16T23:14:11.996233Z","iopub.status.idle":"2026-01-16T23:14:12.001496Z","shell.execute_reply.started":"2026-01-16T23:14:11.996207Z","shell.execute_reply":"2026-01-16T23:14:12.000641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_corpus = train_df[\"lemmatize_text\"][:5].tolist()\nprint(\"The test corpus: \", test_corpus, \"\\n\")\n\ntest_tfidf_em_1gram, test_tfidf_1gram = TFIDF(test_corpus, ngram=1)\nprint_out(test_tfidf_em_1gram, test_tfidf_1gram, ngram=\"Uni-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:15.324546Z","iopub.execute_input":"2026-01-16T23:14:15.324913Z","iopub.status.idle":"2026-01-16T23:14:15.339212Z","shell.execute_reply.started":"2026-01-16T23:14:15.324881Z","shell.execute_reply":"2026-01-16T23:14:15.338444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_tfidf_em_2gram, test_tfidf_2gram = TFIDF(test_corpus, ngram=2)\nprint_out(test_tfidf_em_2gram, test_tfidf_2gram, ngram=\"Bi-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:18.417720Z","iopub.execute_input":"2026-01-16T23:14:18.418078Z","iopub.status.idle":"2026-01-16T23:14:18.430206Z","shell.execute_reply.started":"2026-01-16T23:14:18.418048Z","shell.execute_reply":"2026-01-16T23:14:18.429013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_tfidf_em_3gram, test_tfidf_3gram = TFIDF(test_corpus, ngram=3)\nprint_out(test_tfidf_em_3gram, test_tfidf_3gram, ngram=\"Tri-gram\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:21.404653Z","iopub.execute_input":"2026-01-16T23:14:21.405040Z","iopub.status.idle":"2026-01-16T23:14:21.416462Z","shell.execute_reply.started":"2026-01-16T23:14:21.405002Z","shell.execute_reply":"2026-01-16T23:14:21.415490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\n# implement into the whole dataset\ntrain_df_corpus = train_df[\"lemmatize_text\"].tolist()\ntrain_df_tfidf_1gram, tfidf_1gram = TFIDF(train_df_corpus, 1)\ntrain_df_tfidf_2gram, tfidf_2gram = TFIDF(train_df_corpus, 2)\ntrain_df_tfidf_3gram, tfidf_3gram = TFIDF(train_df_corpus, 3)\n\nprint(len(train_df_corpus))\nprint(train_df_tfidf_1gram.shape)\nprint(train_df_tfidf_1gram.shape)\nprint(train_df_tfidf_1gram.shape)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:24.295774Z","iopub.execute_input":"2026-01-16T23:14:24.296103Z","iopub.status.idle":"2026-01-16T23:14:29.718107Z","shell.execute_reply.started":"2026-01-16T23:14:24.296074Z","shell.execute_reply":"2026-01-16T23:14:29.717312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_df_tfidf_1gram, train_df_tfidf_2gram, train_df_tfidf_3gram","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:29.719669Z","iopub.execute_input":"2026-01-16T23:14:29.719942Z","iopub.status.idle":"2026-01-16T23:14:29.800177Z","shell.execute_reply.started":"2026-01-16T23:14:29.719916Z","shell.execute_reply":"2026-01-16T23:14:29.798904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nimport gensim\nprint(\"gensim version:\", gensim.__version__)\n\nword2vec_path = \"../input/googlenewsvectorsnegative300/GoogleNews-vectors-negative300.bin\"\n\n# we only load 200k most common words from Google News corpus \nword2vec_model = gensim.models.KeyedVectors.load_word2vec_format(word2vec_path, binary=True, limit=200000) ","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:33.881549Z","iopub.execute_input":"2026-01-16T23:14:33.881947Z","iopub.status.idle":"2026-01-16T23:14:36.836860Z","shell.execute_reply.started":"2026-01-16T23:14:33.881914Z","shell.execute_reply":"2026-01-16T23:14:36.836128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compare the similarity between \"cat\" vs. \"kitten\" and \"cat\" vs. \"cats\"","metadata":{}},{"cell_type":"code","source":"print(word2vec_model.similarity('cat', 'kitten'))\nprint(word2vec_model.similarity('cat', 'cats'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:41.788577Z","iopub.execute_input":"2026-01-16T23:14:41.788911Z","iopub.status.idle":"2026-01-16T23:14:41.795338Z","shell.execute_reply.started":"2026-01-16T23:14:41.788880Z","shell.execute_reply":"2026-01-16T23:14:41.794604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_average_vec(tokens_list, vector, generate_missing=False, k=300):\n    \"\"\"\n        Calculate average embedding value of sentence from each word vector\n    \"\"\"\n    \n    if len(tokens_list)<1:\n        return np.zeros(k)\n    \n    if generate_missing:\n        vectorized = [vector[word] if word in vector else np.random.rand(k) for word in tokens_list]\n    else:\n        vectorized = [vector[word] if word in vector else np.zeros(k) for word in tokens_list]\n    \n    length = len(vectorized)\n    summed = np.sum(vectorized, axis=0)\n    averaged = np.divide(summed, length)\n    return averaged\n\ndef get_embeddings(vectors, text, generate_missing=False, k=300):\n    \"\"\"\n        create the sentence embedding\n    \"\"\"\n    embeddings = text.apply(lambda x: get_average_vec(x, vectors, generate_missing=generate_missing, k=k))\n    return list(embeddings)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:42.826181Z","iopub.execute_input":"2026-01-16T23:14:42.826572Z","iopub.status.idle":"2026-01-16T23:14:42.834081Z","shell.execute_reply.started":"2026-01-16T23:14:42.826528Z","shell.execute_reply":"2026-01-16T23:14:42.833356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nembeddings_word2vec = get_embeddings(word2vec_model, train_df[\"lemmatize_text\"], k=300)\n\nprint(\"Embedding matrix size\", len(embeddings_word2vec), len(embeddings_word2vec[0]))\nprint(\"The sentence: \\\"%s\\\" got embedding values: \" % train_df[\"lemmatize_text\"][0])\nprint(embeddings_word2vec[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:45.802007Z","iopub.execute_input":"2026-01-16T23:14:45.802368Z","iopub.status.idle":"2026-01-16T23:14:46.864355Z","shell.execute_reply.started":"2026-01-16T23:14:45.802338Z","shell.execute_reply":"2026-01-16T23:14:46.863391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del embeddings_word2vec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:14:54.881368Z","iopub.execute_input":"2026-01-16T23:14:54.881706Z","iopub.status.idle":"2026-01-16T23:14:54.889197Z","shell.execute_reply.started":"2026-01-16T23:14:54.881677Z","shell.execute_reply":"2026-01-16T23:14:54.888080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nfrom gensim.scripts.glove2word2vec import glove2word2vec\n\nglove_input_file = \"/kaggle/input/glove6b100dtxt/glove.6B.100d.txt\"\nword2vec_output_file = \"glove.6B.100d.txt.word2vec\"\nglove2word2vec(glove_input_file, word2vec_output_file)\n\n# we only load 200k most common words from Google New corpus \nglove_model = gensim.models.KeyedVectors.load_word2vec_format(word2vec_output_file, binary=False, limit=200000) ","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:18:27.298232Z","iopub.execute_input":"2026-01-16T23:18:27.298564Z","iopub.status.idle":"2026-01-16T23:18:50.005482Z","shell.execute_reply.started":"2026-01-16T23:18:27.298531Z","shell.execute_reply":"2026-01-16T23:18:50.004798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compare the similarity between \"cat\" vs. \"kitten\" and \"cat\" vs. \"cats\" from GloVe","metadata":{}},{"cell_type":"code","source":"print(glove_model.similarity('cat', 'kitten'))\nprint(glove_model.similarity('cat', 'cats'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:18:55.615558Z","iopub.execute_input":"2026-01-16T23:18:55.615946Z","iopub.status.idle":"2026-01-16T23:18:55.621553Z","shell.execute_reply.started":"2026-01-16T23:18:55.615911Z","shell.execute_reply":"2026-01-16T23:18:55.620503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nembeddings_glove = get_embeddings(glove_model, train_df[\"lemmatize_text\"], k=100)\n\nprint(\"Embedding matrix size\", len(embeddings_glove), len(embeddings_glove[0]))\nprint(\"The sentence: \\\"%s\\\" got embedding values: \" % train_df[\"lemmatize_text\"][0])\nprint(embeddings_glove[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:20:36.205110Z","iopub.execute_input":"2026-01-16T23:20:36.205449Z","iopub.status.idle":"2026-01-16T23:20:37.114725Z","shell.execute_reply.started":"2026-01-16T23:20:36.205415Z","shell.execute_reply":"2026-01-16T23:20:37.113643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del embeddings_glove","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:20:40.209290Z","iopub.execute_input":"2026-01-16T23:20:40.209592Z","iopub.status.idle":"2026-01-16T23:20:40.216954Z","shell.execute_reply.started":"2026-01-16T23:20:40.209566Z","shell.execute_reply":"2026-01-16T23:20:40.215909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nfrom gensim.models.fasttext import FastText\n\nfasttext_path = \"../input/fasttext-wikinews/wiki-news-300d-1M.vec\"\nfasttext_model = gensim.models.KeyedVectors.load_word2vec_format(fasttext_path, binary=False, limit=200000)","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:20:42.496552Z","iopub.execute_input":"2026-01-16T23:20:42.496885Z","iopub.status.idle":"2026-01-16T23:21:35.990793Z","shell.execute_reply.started":"2026-01-16T23:20:42.496851Z","shell.execute_reply":"2026-01-16T23:21:35.990101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compare the similarity between \"cat\" vs. \"kitten\" and \"cat\" vs. \"cats\" from FastText","metadata":{}},{"cell_type":"code","source":"print(fasttext_model.similarity('cat', 'kitten'))\nprint(fasttext_model.similarity('cat', 'cats'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:21:40.915637Z","iopub.execute_input":"2026-01-16T23:21:40.915941Z","iopub.status.idle":"2026-01-16T23:21:40.922006Z","shell.execute_reply.started":"2026-01-16T23:21:40.915914Z","shell.execute_reply":"2026-01-16T23:21:40.920944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embeddings_fasttext = get_embeddings(fasttext_model, train_df[\"lemmatize_text\"], k=300)\n\nprint(\"Embedding matrix size\", len(embeddings_fasttext), len(embeddings_fasttext[0]))\nprint(\"The sentence: \\\"%s\\\" got embedding values: \" % train_df[\"lemmatize_text\"][0])\nprint(embeddings_fasttext[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:21:43.458226Z","iopub.execute_input":"2026-01-16T23:21:43.458581Z","iopub.status.idle":"2026-01-16T23:21:44.489366Z","shell.execute_reply.started":"2026-01-16T23:21:43.458546Z","shell.execute_reply":"2026-01-16T23:21:44.488441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del embeddings_fasttext","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:21:48.739352Z","iopub.execute_input":"2026-01-16T23:21:48.739671Z","iopub.status.idle":"2026-01-16T23:21:48.746432Z","shell.execute_reply.started":"2026-01-16T23:21:48.739638Z","shell.execute_reply":"2026-01-16T23:21:48.745702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:21:50.800329Z","iopub.execute_input":"2026-01-16T23:21:50.800642Z","iopub.status.idle":"2026-01-16T23:21:50.804474Z","shell.execute_reply.started":"2026-01-16T23:21:50.800610Z","shell.execute_reply":"2026-01-16T23:21:50.803645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install bert-for-tf2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:21:52.554206Z","iopub.execute_input":"2026-01-16T23:21:52.554514Z","iopub.status.idle":"2026-01-16T23:21:58.384066Z","shell.execute_reply.started":"2026-01-16T23:21:52.554487Z","shell.execute_reply":"2026-01-16T23:21:58.383118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time \n\nimport tensorflow_hub as hub\n\n# download the tonkenizer \n!wget --quiet https://raw.githubusercontent.com/tensorflow/models/master/official/nlp/bert/tokenization.py\n","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2026-01-16T23:22:55.009025Z","iopub.execute_input":"2026-01-16T23:22:55.009326Z","iopub.status.idle":"2026-01-16T23:22:56.232717Z","shell.execute_reply.started":"2026-01-16T23:22:55.009301Z","shell.execute_reply":"2026-01-16T23:22:56.231625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"module_url = \"https://tfhub.dev/tensorflow/bert_en_uncased_L-24_H-1024_A-16/1\"\nbert_layer = hub.KerasLayer(module_url, trainable=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:23:00.012687Z","iopub.execute_input":"2026-01-16T23:23:00.013072Z","iopub.status.idle":"2026-01-16T23:23:11.859146Z","shell.execute_reply.started":"2026-01-16T23:23:00.013033Z","shell.execute_reply":"2026-01-16T23:23:11.858150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bert_encode(texts, tokenizer, max_len=512):\n    all_tokens = []\n    all_masks = []\n    all_segments = []\n    \n    for text in texts:\n        text = tokenizer.tokenize(text)\n            \n        text = text[:max_len-2]\n        input_sequence = [\"[CLS]\"] + text + [\"[SEP]\"]\n        pad_len = max_len - len(input_sequence)\n        \n        tokens = tokenizer.convert_tokens_to_ids(input_sequence)\n        tokens += [0] * pad_len\n        pad_masks = [1] * len(input_sequence) + [0] * pad_len\n        segment_ids = [0] * max_len\n        \n        all_tokens.append(tokens)\n        all_masks.append(pad_masks)\n        all_segments.append(segment_ids)\n    \n    return np.array(all_tokens), np.array(all_masks), np.array(all_segments)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:23:15.128611Z","iopub.execute_input":"2026-01-16T23:23:15.128917Z","iopub.status.idle":"2026-01-16T23:23:15.135616Z","shell.execute_reply.started":"2026-01-16T23:23:15.128888Z","shell.execute_reply":"2026-01-16T23:23:15.134879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bert_input = tokenizer.batch_encode_plus(\n    train_df[\"text\"].tolist(),\n    pad_to_max_length=True,\n    truncation=True,\n    max_length=100,\n    return_tensors=\"tf\"\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:27:56.053553Z","iopub.execute_input":"2026-01-16T23:27:56.053896Z","iopub.status.idle":"2026-01-16T23:27:59.498535Z","shell.execute_reply.started":"2026-01-16T23:27:56.053862Z","shell.execute_reply":"2026-01-16T23:27:59.497869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Input IDs tensor shape:\", bert_input[\"input_ids\"].shape)\nprint(\n    'The sentence: \"%s\" got tokenized values:'\n    % train_df[\"lemmatize_text\"].iloc[0]\n)\nprint(\"Input IDs:\")\nprint(bert_input[\"input_ids\"][0])\n\nprint(\"\\nAttention Mask:\")\nprint(bert_input[\"attention_mask\"][0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T23:31:15.295764Z","iopub.execute_input":"2026-01-16T23:31:15.296128Z","iopub.status.idle":"2026-01-16T23:31:15.331625Z","shell.execute_reply.started":"2026-01-16T23:31:15.296100Z","shell.execute_reply":"2026-01-16T23:31:15.330680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}