{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Import Libraries","metadata":{}},{"cell_type":"code","source":"import re\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style(\"whitegrid\")\n\nimport nltk\nimport tensorflow as tf\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom transformers import AutoTokenizer, TFAutoModel\nfrom wordcloud import WordCloud\nfrom sklearn.model_selection import train_test_split\nfrom IPython.display import clear_output\n\nnltk.download('omw-1.4')\nclear_output()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T22:39:44.132634Z","iopub.execute_input":"2022-08-10T22:39:44.133117Z","iopub.status.idle":"2022-08-10T22:39:53.249497Z","shell.execute_reply.started":"2022-08-10T22:39:44.133021Z","shell.execute_reply":"2022-08-10T22:39:53.248130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Loading Dataset","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ndf_test = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:35:54.296281Z","iopub.execute_input":"2022-08-09T23:35:54.296987Z","iopub.status.idle":"2022-08-09T23:35:54.355731Z","shell.execute_reply.started":"2022-08-09T23:35:54.296962Z","shell.execute_reply":"2022-08-09T23:35:54.354745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:35:55.624095Z","iopub.execute_input":"2022-08-09T23:35:55.624544Z","iopub.status.idle":"2022-08-09T23:35:55.645530Z","shell.execute_reply.started":"2022-08-09T23:35:55.624520Z","shell.execute_reply":"2022-08-09T23:35:55.644936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Quick look, we have 5 variables in this train data i.e. Id, keyword, location, text and target. Keyword and location variable seems to have missing values in it. Let's explore our data deeper.","metadata":{}},{"cell_type":"markdown","source":"## 3. Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"### 3.1 Checking Missing Values","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(\n    data=[df_train.isna().sum() / df_train.shape[0] * 100, \n          df_test.isna().sum() / df_test.shape[0] * 100], \n    index=[\"Train Null (%)\", \"Test Null (%)\"]\n).T.style.background_gradient(cmap='summer_r')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:35:57.451991Z","iopub.execute_input":"2022-08-09T23:35:57.452538Z","iopub.status.idle":"2022-08-09T23:35:57.533624Z","shell.execute_reply.started":"2022-08-09T23:35:57.452512Z","shell.execute_reply":"2022-08-09T23:35:57.532291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, we have some missing values on keyword and location variable in both train and test data. Keyword variable only has 0.801% missing value in the train data and 0.797% missing values in the test data. Meanwhile on location variable, we have quite a lot of missing values i.e. 33.27% missing values in the train data and 33.86% missing values in the test data.","metadata":{}},{"cell_type":"markdown","source":"### 3.2 Checking Duplicate Data","metadata":{}},{"cell_type":"code","source":"df_train.loc[df_train[\"text\"].duplicated(keep=False)].sort_values(by=\"text\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:35:59.124387Z","iopub.execute_input":"2022-08-09T23:35:59.124696Z","iopub.status.idle":"2022-08-09T23:35:59.144942Z","shell.execute_reply.started":"2022-08-09T23:35:59.124672Z","shell.execute_reply":"2022-08-09T23:35:59.143940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some train data seems to have duplicate values in the text variable. The interesting thing here is that from some of the duplicate data, there is duplicate data that has a different target value from the other data. We will handle this problem later. For now, we'll just leave it as it is. ","metadata":{}},{"cell_type":"markdown","source":"### 3.3 Checking Target Distribution","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12, 4))\n\ndf_train[\"target\"].value_counts().plot(\n    kind=\"pie\", \n    explode=[0.05 for x in df_train[\"target\"].unique()], \n    autopct='%.2f%%',\n    ax=ax[0], \n    shadow=True \n)\nax[0].set_title(f\"Target Distribution Pie Chart\")\nax[0].set_ylabel('')\n\ncount = sns.countplot(x=\"target\", data=df_train, ax=ax[1])\nfor bar in count.patches:\n    count.annotate(format(bar.get_height()),\n        (bar.get_x() + bar.get_width() / 2,\n        bar.get_height()), ha='center', va='center',\n        size=11, xytext=(0, 8),\n        textcoords='offset points')\nax[1].set_title(f\"Target Distribution Bar Chart\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:00.684385Z","iopub.execute_input":"2022-08-09T23:36:00.684670Z","iopub.status.idle":"2022-08-09T23:36:00.954691Z","shell.execute_reply.started":"2022-08-09T23:36:00.684648Z","shell.execute_reply":"2022-08-09T23:36:00.953771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we have 3271 (42.97%) real disaster tweets and 4243 (57.03%) unreal disaster tweets.","metadata":{}},{"cell_type":"markdown","source":"### 3.4 Checking Tweet Length","metadata":{}},{"cell_type":"markdown","source":"Let's check the number of characters in each tweet.","metadata":{}},{"cell_type":"code","source":"df_train[\"text_length\"] = df_train[\"text\"].apply(lambda x: len(x))\ndf_train.sort_values(by=\"text_length\", ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:04.177026Z","iopub.execute_input":"2022-08-09T23:36:04.177906Z","iopub.status.idle":"2022-08-09T23:36:04.199188Z","shell.execute_reply.started":"2022-08-09T23:36:04.177860Z","shell.execute_reply":"2022-08-09T23:36:04.197944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The longest tweet in the dataset has 157 characters, while the shortest has only 7 characters. Now let's check the tweet length distribution.","metadata":{}},{"cell_type":"code","source":"sns.displot(data=df_train, x=\"text_length\", kde=True, hue=\"target\")\nplt.title(\"Tweet Length Distribution\")\nplt.show()\n\nprint(f\"\\nLowest tweet length: {df_train['text_length'].min()}\")\nprint(f\"Highest tweet length: {df_train['text_length'].max()}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:06.051964Z","iopub.execute_input":"2022-08-09T23:36:06.052297Z","iopub.status.idle":"2022-08-09T23:36:06.732539Z","shell.execute_reply.started":"2022-08-09T23:36:06.052273Z","shell.execute_reply":"2022-08-09T23:36:06.731871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like the character length distribution for both types of tweets is almost the same. The most common number of characters in the tweet is between 130 - 140 characters.\n\nLet's print top 5 longest tweets in the dataset.","metadata":{}},{"cell_type":"code","source":"for tweet in df_train.sort_values(by=\"text_length\", ascending=False)[\"text\"][:5]:\n    print(f\"Tweet: \\n{tweet}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:08.324193Z","iopub.execute_input":"2022-08-09T23:36:08.324930Z","iopub.status.idle":"2022-08-09T23:36:08.333121Z","shell.execute_reply.started":"2022-08-09T23:36:08.324864Z","shell.execute_reply":"2022-08-09T23:36:08.332223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:**\n- Tweets have different casing format.\n- There are so many symbols/special characters in the tweet.\n- Some tweets has mentions.\n- There are some abbreviation.\n- There are duplicate tweets, the only difference is the accounts being mentioned.\n- Some tweets also contains HTML character reference.\n\nWe will deal with this problems later. Now let's count how many special characters in a tweet.","metadata":{}},{"cell_type":"markdown","source":"### 3.5 Checking Tweet Special Characters","metadata":{}},{"cell_type":"code","source":"df_train[\"n_spchars\"] = df_train[\"text\"].map(lambda x: len(re.findall(r\"[\\W_]\", re.sub(r\"\\s\", \"\", x))))\n\nsns.displot(data=df_train, x=\"n_spchars\", kde=True, hue=\"target\", bins=df_train[\"n_spchars\"].nunique())\nplt.title(\"Tweet Special Character Distribution\")\nplt.show()\n\nprint(f\"\\nLowest number of special character in a tweet: {df_train['n_spchars'].min()}\")\nprint(f\"Highest number of special character in a tweet: {df_train['n_spchars'].max()}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:11.052315Z","iopub.execute_input":"2022-08-09T23:36:11.052909Z","iopub.status.idle":"2022-08-09T23:36:11.697681Z","shell.execute_reply.started":"2022-08-09T23:36:11.052864Z","shell.execute_reply":"2022-08-09T23:36:11.696295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the graph above, we know that the majority of tweets have less than 20 special characters. Tweets with the highest number of special characters are known to have a total of 61 special characters.\n\nLet's print top 5 tweets with the highest number of special characters.","metadata":{}},{"cell_type":"code","source":"for tweet in df_train.sort_values(by=\"n_spchars\", ascending=False)[\"text\"][:5]:\n    print(f\"Tweet: \\n{tweet}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:13.624658Z","iopub.execute_input":"2022-08-09T23:36:13.624982Z","iopub.status.idle":"2022-08-09T23:36:13.633267Z","shell.execute_reply.started":"2022-08-09T23:36:13.624958Z","shell.execute_reply":"2022-08-09T23:36:13.632377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:**\n- Some tweets have hashtags\n- There's a url in the tweet","metadata":{}},{"cell_type":"markdown","source":"### 3.6 Checking Number of Words","metadata":{}},{"cell_type":"markdown","source":"(without special characters)","metadata":{}},{"cell_type":"code","source":"df_train[\"n_words\"] = df_train[\"text\"].apply(lambda x: len(re.sub(r\"[\\W_]\", \" \", x).split()))\n\nsns.displot(data=df_train, x=\"n_words\", kde=True, hue=\"target\", bins=df_train[\"n_words\"].nunique())\nplt.title(\"Tweet Word Count Distribution\")\nplt.show()\n\nprint(f\"\\nLowest number of word in a tweet: {df_train['n_words'].min()}\")\nprint(f\"Highest number of word in a tweet: {df_train['n_words'].max()}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:16.124202Z","iopub.execute_input":"2022-08-09T23:36:16.124743Z","iopub.status.idle":"2022-08-09T23:36:16.747473Z","shell.execute_reply.started":"2022-08-09T23:36:16.124712Z","shell.execute_reply":"2022-08-09T23:36:16.746543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the number of words for each tweet in our train data almost have a normal distribution. The majority of tweets in the train data have a word count between 15 - 25. The least number of words in a tweet is 1, while the highest is 34.","metadata":{}},{"cell_type":"code","source":"for tweet in df_train.sort_values(by=\"n_words\", ascending=False)[\"text\"][:5]:\n    print(f\"Tweet: \\n{tweet}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:36:36.877328Z","iopub.execute_input":"2022-08-09T23:36:36.877645Z","iopub.status.idle":"2022-08-09T23:36:36.886957Z","shell.execute_reply.started":"2022-08-09T23:36:36.877620Z","shell.execute_reply":"2022-08-09T23:36:36.885447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:**\n- Nothing new","metadata":{}},{"cell_type":"markdown","source":"### 3.7 Keywords ","metadata":{}},{"cell_type":"code","source":"print(f\" There are {df_train['keyword'].nunique()} unique keywords\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:03.823839Z","iopub.execute_input":"2022-08-09T23:38:03.824676Z","iopub.status.idle":"2022-08-09T23:38:03.829951Z","shell.execute_reply.started":"2022-08-09T23:38:03.824648Z","shell.execute_reply":"2022-08-09T23:38:03.828953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we have 221 unique keywords in our data. Let's check the most common keywords in this data.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\ndf_train[\"keyword\"].value_counts(ascending=True)[-10:].plot.barh()\nplt.title(\"Top 10 Keywords in Train Data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:09.692272Z","iopub.execute_input":"2022-08-09T23:38:09.692593Z","iopub.status.idle":"2022-08-09T23:38:09.878702Z","shell.execute_reply.started":"2022-08-09T23:38:09.692569Z","shell.execute_reply":"2022-08-09T23:38:09.877710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we have fatalities, deluge, and armageddon as the most used keyword in this data. Now let's take a look at the keyword differences between the real disaster tweets and the unreal ones.","metadata":{}},{"cell_type":"code","source":"# replace %20 with a space\ndf_train[\"keyword\"] = df_train[\"keyword\"].str.replace(\"%20\", \" \")\nfig, ax = plt.subplots(1, 2, figsize=(20, 12))\n\nfor i, (target, note) in enumerate(zip([1, 0], [\"Real\", \"Unreal\"])):\n    # count keyword frequency\n    keyword_freq = df_train.loc[df_train[\"target\"]==target, \"keyword\"].value_counts().to_dict()\n    wc = WordCloud().generate_from_frequencies(keyword_freq)\n    ax[i].set_title(f\"{note} Disaster Tweet Keywords\")\n    ax[i].axis(\"off\")\n    ax[i].imshow(wc)\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:21.500784Z","iopub.execute_input":"2022-08-09T23:38:21.501128Z","iopub.status.idle":"2022-08-09T23:38:22.201311Z","shell.execute_reply.started":"2022-08-09T23:38:21.501105Z","shell.execute_reply":"2022-08-09T23:38:22.199950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.8 Location","metadata":{}},{"cell_type":"code","source":"print(f\" There are {df_train['location'].nunique()} unique locations\")\n\nplt.figure(figsize=(8,5))\ndf_train[\"location\"].value_counts(ascending=True)[-10:].plot.barh()\nplt.title(\"Top 10 Locations in Train Data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:32.967860Z","iopub.execute_input":"2022-08-09T23:38:32.968213Z","iopub.status.idle":"2022-08-09T23:38:33.172810Z","shell.execute_reply.started":"2022-08-09T23:38:32.968188Z","shell.execute_reply":"2022-08-09T23:38:33.171612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 3341 unique locations the tweet was sent from, and most tweets are sent from the USA. If you notice, these locations didn't show the same type of location. We can see a country, city, state, and also a combination of them. ","metadata":{}},{"cell_type":"markdown","source":"## 4. Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"Based on the above analysis, we have some findings:\n- There are some duplicated tweets\n- Tweets have different casing format.\n- There are so many symbols/special characters in the tweet.\n- Some tweets has mentions.\n- There are some abbreviation.\n- Some tweets also contains HTML character reference.\n- Some tweets have hashtags.\n- There is a url in the tweet.\n\nNow let's clean our data to make it ready to fit to the classification model.","metadata":{}},{"cell_type":"markdown","source":"### 4.1 Handling Duplicate Data ","metadata":{}},{"cell_type":"markdown","source":"As we know from the previous analysis, some duplicate data have different target values. After observing the data, we decide to delete the duplicate data and keep the first data because the first data seems to have the right target value.","metadata":{}},{"cell_type":"code","source":"print(f\"Number of train data before drop duplicates: {df_train.shape[0]}\")\ndf_train = df_train.drop_duplicates(subset=\"text\", keep=\"first\").reset_index(drop=True)\nprint(f\"Number of train data after drop duplicates: {df_train.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:40.464781Z","iopub.execute_input":"2022-08-09T23:38:40.465097Z","iopub.status.idle":"2022-08-09T23:38:40.476137Z","shell.execute_reply.started":"2022-08-09T23:38:40.465074Z","shell.execute_reply":"2022-08-09T23:38:40.474788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.2 Lowercasing Data","metadata":{}},{"cell_type":"markdown","source":"We will convert the text to lowercase so all data will have the same casing format. ","metadata":{}},{"cell_type":"code","source":"# we will use the text of the data with id = 885 as an example\ntweet = df_train.loc[df_train[\"id\"]==885, \"text\"].values[0]\nprint(f\"Before:\\n{tweet}\\n\")\n\ntweet = tweet.lower()\nprint(f\"After:\\n{tweet}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:42.997479Z","iopub.execute_input":"2022-08-09T23:38:42.997907Z","iopub.status.idle":"2022-08-09T23:38:43.005826Z","shell.execute_reply.started":"2022-08-09T23:38:42.997855Z","shell.execute_reply":"2022-08-09T23:38:43.004441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.3 Remove Mentions","metadata":{}},{"cell_type":"markdown","source":"Next step is removing mention because this part of the tweet is useless for the classification model.","metadata":{}},{"cell_type":"code","source":"print(f\"Before:\\n{tweet}\\n\")\n\ntweet = re.sub(r\"@\\S+\", \" \", tweet)\nprint(f\"After:\\n{tweet}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:44.472032Z","iopub.execute_input":"2022-08-09T23:38:44.473353Z","iopub.status.idle":"2022-08-09T23:38:44.478537Z","shell.execute_reply.started":"2022-08-09T23:38:44.473307Z","shell.execute_reply":"2022-08-09T23:38:44.477490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.4 Remove HTML Character Reference and Tags","metadata":{}},{"cell_type":"markdown","source":"We will also remove another useless part, which is HTML character reference and HTML tags","metadata":{}},{"cell_type":"code","source":"print(f\"Before:\\n{tweet}\\n\")\n\ntweet = re.sub(r\"&.*?;|<.*?>\", \" \", tweet)\nprint(f\"After:\\n{tweet}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:46.600233Z","iopub.execute_input":"2022-08-09T23:38:46.600590Z","iopub.status.idle":"2022-08-09T23:38:46.606747Z","shell.execute_reply.started":"2022-08-09T23:38:46.600564Z","shell.execute_reply":"2022-08-09T23:38:46.605452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.5 Remove URL","metadata":{}},{"cell_type":"markdown","source":"And again, another useless part that is url.","metadata":{}},{"cell_type":"code","source":"# we will use the text of the data with id = 520 as an example since there's no url in the previous text\ntweet2 = df_train.loc[df_train[\"id\"]==520, \"text\"].values[0]\nprint(f\"Before:\\n{tweet2}\\n\")\n\ntweet2 = re.sub(r\"https?://\\S+|www\\.\\S+\", \" \", tweet2)\nprint(f\"After:\\n{tweet2}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:49.425254Z","iopub.execute_input":"2022-08-09T23:38:49.426328Z","iopub.status.idle":"2022-08-09T23:38:49.432961Z","shell.execute_reply.started":"2022-08-09T23:38:49.426299Z","shell.execute_reply":"2022-08-09T23:38:49.431657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.6 Abbreviation Handling","metadata":{}},{"cell_type":"markdown","source":"As we know from above analysis, we have some abbreviations in this data. So in this step, we will convert the abbreviation into full words. \n\nThanks to https://www.kaggle.com/rftexas/text-only-kfold-bert","metadata":{}},{"cell_type":"code","source":"abbreviations = {\n    \"$\" : \" dollar \",\n    \"€\" : \" euro \",\n    \"4ao\" : \"for adults only\",\n    \"a.m\" : \"before midday\",\n    \"a3\" : \"anytime anywhere anyplace\",\n    \"aamof\" : \"as a matter of fact\",\n    \"acct\" : \"account\",\n    \"adih\" : \"another day in hell\",\n    \"afaic\" : \"as far as i am concerned\",\n    \"afaict\" : \"as far as i can tell\",\n    \"afaik\" : \"as far as i know\",\n    \"afair\" : \"as far as i remember\",\n    \"afk\" : \"away from keyboard\",\n    \"app\" : \"application\",\n    \"approx\" : \"approximately\",\n    \"apps\" : \"applications\",\n    \"asap\" : \"as soon as possible\",\n    \"asl\" : \"age, sex, location\",\n    \"atk\" : \"at the keyboard\",\n    \"ave.\" : \"avenue\",\n    \"aymm\" : \"are you my mother\",\n    \"ayor\" : \"at your own risk\", \n    \"b&b\" : \"bed and breakfast\",\n    \"b+b\" : \"bed and breakfast\",\n    \"b.c\" : \"before christ\",\n    \"b2b\" : \"business to business\",\n    \"b2c\" : \"business to customer\",\n    \"b4\" : \"before\",\n    \"b4n\" : \"bye for now\",\n    \"b@u\" : \"back at you\",\n    \"bae\" : \"before anyone else\",\n    \"bak\" : \"back at keyboard\",\n    \"bbbg\" : \"bye bye be good\",\n    \"bbc\" : \"british broadcasting corporation\",\n    \"bbias\" : \"be back in a second\",\n    \"bbl\" : \"be back later\",\n    \"bbs\" : \"be back soon\",\n    \"be4\" : \"before\",\n    \"bfn\" : \"bye for now\",\n    \"blvd\" : \"boulevard\",\n    \"bout\" : \"about\",\n    \"brb\" : \"be right back\",\n    \"bros\" : \"brothers\",\n    \"brt\" : \"be right there\",\n    \"bsaaw\" : \"big smile and a wink\",\n    \"btw\" : \"by the way\",\n    \"bwl\" : \"bursting with laughter\",\n    \"c/o\" : \"care of\",\n    \"cet\" : \"central european time\",\n    \"cf\" : \"compare\",\n    \"cia\" : \"central intelligence agency\",\n    \"csl\" : \"can not stop laughing\",\n    \"cu\" : \"see you\",\n    \"cul8r\" : \"see you later\",\n    \"cv\" : \"curriculum vitae\",\n    \"cwot\" : \"complete waste of time\",\n    \"cya\" : \"see you\",\n    \"cyt\" : \"see you tomorrow\",\n    \"dae\" : \"does anyone else\",\n    \"dbmib\" : \"do not bother me i am busy\",\n    \"diy\" : \"do it yourself\",\n    \"dm\" : \"direct message\",\n    \"dwh\" : \"during work hours\",\n    \"e123\" : \"easy as one two three\",\n    \"eet\" : \"eastern european time\",\n    \"eg\" : \"example\",\n    \"embm\" : \"early morning business meeting\",\n    \"encl\" : \"enclosed\",\n    \"encl.\" : \"enclosed\",\n    \"etc\" : \"and so on\",\n    \"faq\" : \"frequently asked questions\",\n    \"fawc\" : \"for anyone who cares\",\n    \"fb\" : \"facebook\",\n    \"fc\" : \"fingers crossed\",\n    \"fig\" : \"figure\",\n    \"fimh\" : \"forever in my heart\", \n    \"ft.\" : \"feet\",\n    \"ft\" : \"featuring\",\n    \"ftl\" : \"for the loss\",\n    \"ftw\" : \"for the win\",\n    \"fwiw\" : \"for what it is worth\",\n    \"fyi\" : \"for your information\",\n    \"g9\" : \"genius\",\n    \"gahoy\" : \"get a hold of yourself\",\n    \"gal\" : \"get a life\",\n    \"gcse\" : \"general certificate of secondary education\",\n    \"gfn\" : \"gone for now\",\n    \"gg\" : \"good game\",\n    \"gl\" : \"good luck\",\n    \"glhf\" : \"good luck have fun\",\n    \"gmt\" : \"greenwich mean time\",\n    \"gmta\" : \"great minds think alike\",\n    \"gn\" : \"good night\",\n    \"g.o.a.t\" : \"greatest of all time\",\n    \"goat\" : \"greatest of all time\",\n    \"goi\" : \"get over it\",\n    \"gps\" : \"global positioning system\",\n    \"gr8\" : \"great\",\n    \"gratz\" : \"congratulations\",\n    \"gyal\" : \"girl\",\n    \"h&c\" : \"hot and cold\",\n    \"hp\" : \"horsepower\",\n    \"hr\" : \"hour\",\n    \"hrh\" : \"his royal highness\",\n    \"ht\" : \"height\",\n    \"ibrb\" : \"i will be right back\",\n    \"ic\" : \"i see\",\n    \"icq\" : \"i seek you\",\n    \"icymi\" : \"in case you missed it\",\n    \"idc\" : \"i do not care\",\n    \"idgadf\" : \"i do not give a damn fuck\",\n    \"idgaf\" : \"i do not give a fuck\",\n    \"idk\" : \"i do not know\",\n    \"ie\" : \"that is\",\n    \"i.e\" : \"that is\",\n    \"ifyp\" : \"i feel your pain\",\n    \"ig\" : \"instagram\",\n    \"iirc\" : \"if i remember correctly\",\n    \"ilu\" : \"i love you\",\n    \"ily\" : \"i love you\",\n    \"imho\" : \"in my humble opinion\",\n    \"imo\" : \"in my opinion\",\n    \"imu\" : \"i miss you\",\n    \"iow\" : \"in other words\",\n    \"irl\" : \"in real life\",\n    \"j4f\" : \"just for fun\",\n    \"jic\" : \"just in case\",\n    \"jk\" : \"just kidding\",\n    \"jsyk\" : \"just so you know\",\n    \"l8r\" : \"later\",\n    \"lb\" : \"pound\",\n    \"lbs\" : \"pounds\",\n    \"ldr\" : \"long distance relationship\",\n    \"lmao\" : \"laugh my ass off\",\n    \"lmfao\" : \"laugh my fucking ass off\",\n    \"lol\" : \"laughing out loud\",\n    \"ltd\" : \"limited\",\n    \"ltns\" : \"long time no see\",\n    \"m8\" : \"mate\",\n    \"mf\" : \"motherfucker\",\n    \"mfs\" : \"motherfuckers\",\n    \"mfw\" : \"my face when\",\n    \"mofo\" : \"motherfucker\",\n    \"mph\" : \"miles per hour\",\n    \"mr\" : \"mister\",\n    \"mrw\" : \"my reaction when\",\n    \"ms\" : \"miss\",\n    \"mte\" : \"my thoughts exactly\",\n    \"nagi\" : \"not a good idea\",\n    \"nbc\" : \"national broadcasting company\",\n    \"nbd\" : \"not big deal\",\n    \"nfs\" : \"not for sale\",\n    \"ngl\" : \"not going to lie\",\n    \"nhs\" : \"national health service\",\n    \"nrn\" : \"no reply necessary\",\n    \"nsfl\" : \"not safe for life\",\n    \"nsfw\" : \"not safe for work\",\n    \"nth\" : \"nice to have\",\n    \"nvr\" : \"never\",\n    \"nyc\" : \"new york city\",\n    \"oc\" : \"original content\",\n    \"og\" : \"original\",\n    \"ohp\" : \"overhead projector\",\n    \"oic\" : \"oh i see\",\n    \"omdb\" : \"over my dead body\",\n    \"omg\" : \"oh my god\",\n    \"omw\" : \"on my way\",\n    \"p.a\" : \"per annum\",\n    \"p.m\" : \"after midday\",\n    \"pm\" : \"prime minister\",\n    \"poc\" : \"people of color\",\n    \"pov\" : \"point of view\",\n    \"pp\" : \"pages\",\n    \"ppl\" : \"people\",\n    \"prw\" : \"parents are watching\",\n    \"ps\" : \"postscript\",\n    \"pt\" : \"point\",\n    \"ptb\" : \"please text back\",\n    \"pto\" : \"please turn over\",\n    \"qpsa\" : \"what happens\", #\"que pasa\",\n    \"ratchet\" : \"rude\",\n    \"rbtl\" : \"read between the lines\",\n    \"rlrt\" : \"real life retweet\", \n    \"rofl\" : \"rolling on the floor laughing\",\n    \"roflol\" : \"rolling on the floor laughing out loud\",\n    \"rotflmao\" : \"rolling on the floor laughing my ass off\",\n    \"rt\" : \"retweet\",\n    \"ruok\" : \"are you ok\",\n    \"sfw\" : \"safe for work\",\n    \"sk8\" : \"skate\",\n    \"smh\" : \"shake my head\",\n    \"sq\" : \"square\",\n    \"srsly\" : \"seriously\", \n    \"ssdd\" : \"same stuff different day\",\n    \"tbh\" : \"to be honest\",\n    \"tbs\" : \"tablespooful\",\n    \"tbsp\" : \"tablespooful\",\n    \"tfw\" : \"that feeling when\",\n    \"thks\" : \"thank you\",\n    \"tho\" : \"though\",\n    \"thx\" : \"thank you\",\n    \"tia\" : \"thanks in advance\",\n    \"til\" : \"today i learned\",\n    \"tl;dr\" : \"too long i did not read\",\n    \"tldr\" : \"too long i did not read\",\n    \"tmb\" : \"tweet me back\",\n    \"tntl\" : \"trying not to laugh\",\n    \"ttyl\" : \"talk to you later\",\n    \"u\" : \"you\",\n    \"u2\" : \"you too\",\n    \"u4e\" : \"yours for ever\",\n    \"utc\" : \"coordinated universal time\",\n    \"w/\" : \"with\",\n    \"w/o\" : \"without\",\n    \"w8\" : \"wait\",\n    \"wassup\" : \"what is up\",\n    \"wb\" : \"welcome back\",\n    \"wtf\" : \"what the fuck\",\n    \"wtg\" : \"way to go\",\n    \"wtpa\" : \"where the party at\",\n    \"wuf\" : \"where are you from\",\n    \"wuzup\" : \"what is up\",\n    \"wywh\" : \"wish you were here\",\n    \"yd\" : \"yard\",\n    \"ygtr\" : \"you got that right\",\n    \"ynk\" : \"you never know\",\n    \"zzz\" : \"sleeping bored and tired\"\n}","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-10T22:40:24.257139Z","iopub.execute_input":"2022-08-10T22:40:24.257937Z","iopub.status.idle":"2022-08-10T22:40:24.286026Z","shell.execute_reply.started":"2022-08-10T22:40:24.257893Z","shell.execute_reply":"2022-08-10T22:40:24.284987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_abbrev(word):\n    return abbreviations[word.lower()] if word.lower() in abbreviations.keys() else word\n\ndef convert_abbrev_in_text(text):\n    tokens = word_tokenize(text)\n    tokens = [convert_abbrev(word) for word in tokens]\n    text = ' '.join(tokens)\n    \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-10T22:40:25.949237Z","iopub.execute_input":"2022-08-10T22:40:25.950376Z","iopub.status.idle":"2022-08-10T22:40:25.957484Z","shell.execute_reply.started":"2022-08-10T22:40:25.950289Z","shell.execute_reply":"2022-08-10T22:40:25.956129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.7 Remove Non-Word Character","metadata":{}},{"cell_type":"markdown","source":"We only need letter data to create a classification model. Therefore, we will remove all numbers, symbols and special characters in the text.","metadata":{}},{"cell_type":"code","source":"print(f\"Before:\\n{tweet}\\n\")\n\ntweet = re.sub(r\"[^a-z]\", \" \", tweet)\nprint(f\"After:\\n{tweet}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:38:56.040353Z","iopub.execute_input":"2022-08-09T23:38:56.040875Z","iopub.status.idle":"2022-08-09T23:38:56.046251Z","shell.execute_reply.started":"2022-08-09T23:38:56.040849Z","shell.execute_reply":"2022-08-09T23:38:56.044980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.8 Remove Stopwords","metadata":{}},{"cell_type":"markdown","source":"Stop words are a set of commonly used words in a language. By removing these words, we remove the low-level information from our text in order to give more focus to the important information.","metadata":{}},{"cell_type":"code","source":"# we change our example text to give a clearer example\ntweet3 = df_train.loc[df_train[\"id\"]==5, \"text\"].values[0]\nprint(f\"Before:\\n{tweet3}\\n\")\n\ntweet3 = \" \".join(word for word in word_tokenize(tweet3) if word not in stopwords.words('english'))\nprint(f\"After:\\n{tweet3}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:39:03.496229Z","iopub.execute_input":"2022-08-09T23:39:03.496561Z","iopub.status.idle":"2022-08-09T23:39:03.526850Z","shell.execute_reply.started":"2022-08-09T23:39:03.496536Z","shell.execute_reply":"2022-08-09T23:39:03.526225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.9 Lemmatization","metadata":{}},{"cell_type":"markdown","source":"Lemmatization is the process of grouping together the different inflected forms of a word with the use of a vocabulary and morphological analysis of words, normally aiming to remove inflectional endings only and to return the base or dictionary form of a word, so they can be analyzed as a single item. In his case, we will use WordNetLemmatizer from nltk to do this task.","metadata":{}},{"cell_type":"code","source":"lemma = WordNetLemmatizer()\nprint(f\"Before:\\n{tweet3}\\n\")\n\ntweet3 = \" \".join(lemma.lemmatize(word) for word in word_tokenize(tweet3))\nprint(f\"After:\\n{tweet3}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:39:10.164678Z","iopub.execute_input":"2022-08-09T23:39:10.165014Z","iopub.status.idle":"2022-08-09T23:39:11.865303Z","shell.execute_reply.started":"2022-08-09T23:39:10.164989Z","shell.execute_reply":"2022-08-09T23:39:11.863757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.10 Remove Single Character","metadata":{}},{"cell_type":"markdown","source":"The last step is remove any single character in the text because it provides very low information.","metadata":{}},{"cell_type":"code","source":"print(f\"Before:\\n{tweet}\\n\")\n\ntweet = re.sub(r\"\\b\\w\\b\", \" \", tweet)\nprint(f\"After:\\n{tweet}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:39:22.445689Z","iopub.execute_input":"2022-08-09T23:39:22.446228Z","iopub.status.idle":"2022-08-09T23:39:22.452394Z","shell.execute_reply.started":"2022-08-09T23:39:22.446203Z","shell.execute_reply":"2022-08-09T23:39:22.451114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.11 Putting All Data Cleaning Techniques Together","metadata":{}},{"cell_type":"markdown","source":"Finally, let's gather our data cleaning techniques in a function for better reusability.","metadata":{}},{"cell_type":"code","source":"def clean_text(tweet):\n    # convert tweet to lower case\n    tweet = tweet.lower()\n    \n    # remove mentions\n    tweet = re.sub(r\"@\\S+\", \" \", tweet)\n    \n    # remove html character reference and tags\n    tweet = re.sub(r\"&.*?;|<.*?>\", \" \", tweet)\n    \n    # remove url\n    tweet = re.sub(r\"https?://\\S+|www\\.\\S+\", \" \", tweet)\n    \n    # abbreviation handling\n    tweet = convert_abbrev_in_text(tweet)\n    \n    # remove non-word characters\n    tweet = re.sub(r\"[^a-z]\", \" \", tweet)\n    \n    # remove stopwords\n    tweet = \" \".join(word for word in word_tokenize(tweet) if word not in stopwords.words('english'))\n    \n    # lemmatization\n    tweet = \" \".join(lemma.lemmatize(word) for word in word_tokenize(tweet))\n    \n    # remove single characters\n    tweet = re.sub(r\"\\b\\w\\b\", \"\", tweet).strip()\n    \n    return tweet","metadata":{"execution":{"iopub.status.busy":"2022-08-10T22:41:28.468907Z","iopub.execute_input":"2022-08-10T22:41:28.470075Z","iopub.status.idle":"2022-08-10T22:41:28.480207Z","shell.execute_reply.started":"2022-08-10T22:41:28.470020Z","shell.execute_reply":"2022-08-10T22:41:28.478631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"text_clean\"] = df_train[\"text\"].apply(clean_text)\ndf_test[\"text_clean\"] = df_test[\"text\"].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:39:39.037780Z","iopub.execute_input":"2022-08-09T23:39:39.038128Z","iopub.status.idle":"2022-08-09T23:39:59.676714Z","shell.execute_reply.started":"2022-08-09T23:39:39.038103Z","shell.execute_reply":"2022-08-09T23:39:59.675457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"### 5.1 Checking Number of Words After Data Cleaning","metadata":{}},{"cell_type":"code","source":"df_train[\"n_words_clean\"] = df_train[\"text_clean\"].apply(lambda x: len(re.sub(r\"[\\W_]\", \" \", x).split()))\n\nsns.displot(data=df_train, x=\"n_words_clean\", kde=True, hue=\"target\", bins=df_train[\"n_words_clean\"].nunique())\nplt.title(\"Tweet Word Count Distribution After Cleaning\")\nplt.show()\n\nprint(f\"\\nLowest number of word in a tweet: {df_train['n_words_clean'].min()}\")\nprint(f\"Highest number of word in a tweet: {df_train['n_words_clean'].max()}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:40:12.441551Z","iopub.execute_input":"2022-08-09T23:40:12.441850Z","iopub.status.idle":"2022-08-09T23:40:12.980602Z","shell.execute_reply.started":"2022-08-09T23:40:12.441826Z","shell.execute_reply":"2022-08-09T23:40:12.978381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow, we have tweet with 0 word now after data cleaning. Let's check which one is the data.","metadata":{}},{"cell_type":"code","source":"display(df_train[df_train.n_words_clean == 0])\n\nprint(\"\\nOriginal text:\")\nfor i, text in enumerate(df_train[df_train.n_words_clean == 0][\"text\"].values):\n    print(f\"  {i+1}. {text}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:40:58.221399Z","iopub.execute_input":"2022-08-09T23:40:58.221741Z","iopub.status.idle":"2022-08-09T23:40:58.241404Z","shell.execute_reply.started":"2022-08-09T23:40:58.221710Z","shell.execute_reply":"2022-08-09T23:40:58.240309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Apparently, we have 2 blank tweets after data cleaning. As we can see, the text only contains mentions, special cahracters, and stopwords. They have been removed during the cleaning process. Is there a way to fill in the blank text? Yes, keywords. We can add keyword to the text because keyword might be useful since it represent the content of the text itself.","metadata":{}},{"cell_type":"markdown","source":"### 5.2 Adding Keyword To Text","metadata":{}},{"cell_type":"code","source":"for df in [df_train, df_test]:\n    df[\"keyword\"] = df[\"keyword\"].str.replace(\"%20\", \" \") # replace %20 with space\n    df[\"keyword\"].fillna(\"\", inplace=True) # fill missing values with blank string\n    df[\"text_clean\"] = df[\"text_clean\"] + \" \" + df[\"keyword\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:42:54.444286Z","iopub.execute_input":"2022-08-09T23:42:54.444608Z","iopub.status.idle":"2022-08-09T23:42:54.464089Z","shell.execute_reply.started":"2022-08-09T23:42:54.444584Z","shell.execute_reply":"2022-08-09T23:42:54.462692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.3 Checking Duplicate Data After Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"We know that some tweets contains exactly the same text, but different people are mentioned in those tweets (check 3.4). This makes the tweets not detected as duplicate data in the previous process. After we have done the data cleaning process to deal with useless data including mentions, now we should be able to find duplicate tweets that were previously undetected. Let's check how many duplicate data we have.","metadata":{}},{"cell_type":"code","source":"print(f\"There are {df_train['text_clean'].duplicated().sum()} duplicate text in train data\\n\")\ndf_train.loc[df_train[\"text_clean\"].duplicated(keep=False)].sort_values(\"text_clean\").head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:42:58.843313Z","iopub.execute_input":"2022-08-09T23:42:58.843620Z","iopub.status.idle":"2022-08-09T23:42:58.862669Z","shell.execute_reply.started":"2022-08-09T23:42:58.843597Z","shell.execute_reply":"2022-08-09T23:42:58.861755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow, we still have 715 duplicate data. Let's drop them again.","metadata":{}},{"cell_type":"code","source":"print(f\"Number of train data before drop duplicates: {df_train.shape[0]}\")\ndf_train = df_train.drop_duplicates(subset=\"text_clean\", keep=\"first\").reset_index(drop=True)\nprint(f\"Number of train data after drop duplicates: {df_train.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-09T23:43:12.779191Z","iopub.execute_input":"2022-08-09T23:43:12.779481Z","iopub.status.idle":"2022-08-09T23:43:12.789464Z","shell.execute_reply.started":"2022-08-09T23:43:12.779459Z","shell.execute_reply":"2022-08-09T23:43:12.788471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.4 Splitting Dataset","metadata":{}},{"cell_type":"markdown","source":"We will split our dataset into 90% train set and 10% validation set.","metadata":{}},{"cell_type":"code","source":"X = df_train[\"text_clean\"]\ny = df_train[\"target\"]\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=0)\nX_test = df_test[\"text_clean\"]\n\nfor (x, name) in zip([X_train, X_val, X_test], [\"train\", \"validation\", \"test\"]):\n    print(f\"Number of {name} data: {x.size}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:13:40.991331Z","iopub.execute_input":"2022-08-11T01:13:40.991756Z","iopub.status.idle":"2022-08-11T01:13:41.002628Z","shell.execute_reply.started":"2022-08-11T01:13:40.991724Z","shell.execute_reply":"2022-08-11T01:13:41.001362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.5 Text Encoding","metadata":{}},{"cell_type":"markdown","source":"Text encoding is a process to convert meaningful text into number / vector representation so as to preserve the context and relationship between words and sentences, such that a machine can understand the pattern associated in any text and can make out the context of sentences. \n\nThere are several text encoding methods that we can use, but in this case we will use BERT (Bidirectional Encoder Representations from Transformers). ","metadata":{}},{"cell_type":"code","source":"seq_length = 23 # initialize sequence length to 23 (23 is our max number of words after data cleaning)\nbatch_size = 16 # initialize batch size to 16\n\n# initialize tokenizer\ntokenizer = AutoTokenizer.from_pretrained(\"bert-base-uncased\")\n\n# create a function to encode our text data\ndef tokenize(X, max_length=seq_length, tokenizer=tokenizer):\n    tokens = tokenizer.batch_encode_plus(X.to_list(),\n                                   padding=\"max_length\",\n                                   max_length=max_length,\n                                   truncation=True,\n                                   return_attention_mask=True,\n                                   return_token_type_ids=False,\n                                   return_tensors='tf'\n                                  )\n    \n    return tokens[\"input_ids\"], tokens[\"attention_mask\"]\n\n# create a mapping function to restructure our dataset\ndef map_func(input_ids, masks, labels):\n    return {\"input_ids\": input_ids, \"attention_mask\": masks}, labels\n\n# create a function to preprocess our dataset\ndef preprocess(X, y, batch_size=batch_size):\n    X_ids, X_mask = tokenize(X) # tokenize (encode) dataset\n    labels = pd.get_dummies(y) # convert label/target 1D -> 2D\n    dataset = tf.data.Dataset.from_tensor_slices((X_ids, X_mask, labels)) # load data into tensorflow dataset\n    dataset = dataset.map(map_func).batch(batch_size) # restructure and batch dataset\n    \n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:52:24.059207Z","iopub.execute_input":"2022-08-11T01:52:24.059660Z","iopub.status.idle":"2022-08-11T01:52:25.007432Z","shell.execute_reply.started":"2022-08-11T01:52:24.059628Z","shell.execute_reply":"2022-08-11T01:52:25.005923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = preprocess(X_train, y_train)\nval_ds = preprocess(X_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T01:56:23.959507Z","iopub.execute_input":"2022-08-11T01:56:23.959979Z","iopub.status.idle":"2022-08-11T01:56:24.285825Z","shell.execute_reply.started":"2022-08-11T01:56:23.959941Z","shell.execute_reply":"2022-08-11T01:56:24.284668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Model Building","metadata":{}},{"cell_type":"code","source":"from transformers import TFAutoModel\nbert = TFAutoModel.from_pretrained(\"bert-base-uncased\")\n\ninput_ids = tf.keras.layers.Input(shape=(seq_length,), name='input_ids', dtype='int32')\nmask = tf.keras.layers.Input(shape=(seq_length,), name='attention_mask', dtype='int32')\n\nembeddings = bert(input_ids, attention_mask=mask)[0]\nX = tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(32, dropout=0.4, recurrent_dropout=0.4, activation='relu'))(embeddings)\nX = tf.keras.layers.Dense(128, activation='relu')(X)\nX = tf.keras.layers.Dropout(0.3)(X)\nX = tf.keras.layers.Dense(64, activation='relu')(X)\nX = tf.keras.layers.Dropout(0.2)(X)\ny = tf.keras.layers.Dense(2, activation='sigmoid', name='outputs')(X)\n\nmodel = tf.keras.Model(inputs=[input_ids, mask], outputs=y)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:20:24.622132Z","iopub.execute_input":"2022-08-11T02:20:24.622984Z","iopub.status.idle":"2022-08-11T02:20:29.887710Z","shell.execute_reply.started":"2022-08-11T02:20:24.622936Z","shell.execute_reply":"2022-08-11T02:20:29.886502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(2e-5)\nloss = tf.keras.losses.CategoricalCrossentropy()\n\nmodel.compile(optimizer=optimizer, loss=loss, metrics=['accuracy'])\nmodel.fit(train_ds, validation_data=val_ds, epochs=2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:21:32.033335Z","iopub.execute_input":"2022-08-11T02:21:32.034271Z","iopub.status.idle":"2022-08-11T02:51:59.464205Z","shell.execute_reply.started":"2022-08-11T02:21:32.034218Z","shell.execute_reply":"2022-08-11T02:51:59.462719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making Submission","metadata":{}},{"cell_type":"code","source":"def predict(X):\n    X_ids, X_mask = tokenize(X)\n    pred = model.predict((X_ids, X_mask))\n    \n    return np.argmax(pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:51:59.615773Z","iopub.execute_input":"2022-08-11T02:51:59.616572Z","iopub.status.idle":"2022-08-11T02:54:10.703600Z","shell.execute_reply.started":"2022-08-11T02:51:59.616537Z","shell.execute_reply":"2022-08-11T02:54:10.702197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = predict(X_test)\n\nsubmission = pd.DataFrame({\n    'id': df_test.id, \n    'target': y_pred\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:09:22.936097Z","iopub.execute_input":"2022-08-11T03:09:22.936535Z","iopub.status.idle":"2022-08-11T03:11:27.489063Z","shell.execute_reply.started":"2022-08-11T03:09:22.936502Z","shell.execute_reply":"2022-08-11T03:11:27.487740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}