{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\n<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>1.1 Objectives</b></p>\n</div>\n\nI'm very excited to participate in kaggle's Getting Started with **NLP**.\nIn this competition, we’re challenged to build a machine learning model that predicts which Tweets are about real disasters and which one’s aren’t. \n\nWe have access to a dataset of 10,000 tweets that were hand classified. \n\nIn this notebook, I've performed below action items:\n* EDA of data to get insight \n* Preprocessing of text\n* implementation of advance version of embedding from tf-hub\n* build a dense layerd model","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>1.2 Loading Libraries</b></p>\n</div>\n","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re\nfrom sklearn import model_selection\nfrom IPython.display import display\nfrom collections import defaultdict\nfrom collections import  Counter\n\n# plotting\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import TextVectorization\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom tqdm.notebook import tqdm_notebook\n\nfrom wordcloud import STOPWORDS, WordCloud\nfrom termcolor import colored\n\n\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\ntf.__version__","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:21.587452Z","iopub.execute_input":"2022-07-13T17:37:21.588451Z","iopub.status.idle":"2022-07-13T17:37:27.735712Z","shell.execute_reply.started":"2022-07-13T17:37:21.588359Z","shell.execute_reply":"2022-07-13T17:37:27.734600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Data Overview \n\n\n\n<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>2.1 Loading the Data</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ndf_test = pd.read_csv(\"../input/nlp-getting-started/test.csv\")\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.737546Z","iopub.execute_input":"2022-07-13T17:37:27.738474Z","iopub.status.idle":"2022-07-13T17:37:27.833849Z","shell.execute_reply.started":"2022-07-13T17:37:27.738436Z","shell.execute_reply":"2022-07-13T17:37:27.832997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('There are {} rows and {} columns in train'.format(df_train.shape[0], df_train.shape[1]))\nprint('There are {} rows and {} columns in train'.format(df_test.shape[0],df_test.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.836702Z","iopub.execute_input":"2022-07-13T17:37:27.836965Z","iopub.status.idle":"2022-07-13T17:37:27.843485Z","shell.execute_reply.started":"2022-07-13T17:37:27.836942Z","shell.execute_reply":"2022-07-13T17:37:27.842499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.846163Z","iopub.execute_input":"2022-07-13T17:37:27.848116Z","iopub.status.idle":"2022-07-13T17:37:27.871689Z","shell.execute_reply.started":"2022-07-13T17:37:27.848078Z","shell.execute_reply":"2022-07-13T17:37:27.870722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>2.2  Summary Statistics</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"df_train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.872998Z","iopub.execute_input":"2022-07-13T17:37:27.873324Z","iopub.status.idle":"2022-07-13T17:37:27.898160Z","shell.execute_reply.started":"2022-07-13T17:37:27.873291Z","shell.execute_reply":"2022-07-13T17:37:27.897344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe(include=\"O\").T","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.899144Z","iopub.execute_input":"2022-07-13T17:37:27.899379Z","iopub.status.idle":"2022-07-13T17:37:27.926532Z","shell.execute_reply.started":"2022-07-13T17:37:27.899357Z","shell.execute_reply":"2022-07-13T17:37:27.925716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[df_train.text.str.contains(\"Boy Charged\")][[\"text\", \"target\", \"location\"]]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.929112Z","iopub.execute_input":"2022-07-13T17:37:27.929350Z","iopub.status.idle":"2022-07-13T17:37:27.948600Z","shell.execute_reply.started":"2022-07-13T17:37:27.929328Z","shell.execute_reply":"2022-07-13T17:37:27.947746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Removing the duplicate texts\ndf_train.drop_duplicates(subset=[\"text\", \"location\", \"target\"], inplace=True)\nprint('There are {} rows and {} columns in train'.format(df_train.shape[0], df_train.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.950629Z","iopub.execute_input":"2022-07-13T17:37:27.951011Z","iopub.status.idle":"2022-07-13T17:37:27.963966Z","shell.execute_reply.started":"2022-07-13T17:37:27.950972Z","shell.execute_reply":"2022-07-13T17:37:27.963012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>2.3 Target Class Distribution</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.subplot(1, 2, 1)\ndf_train.target.value_counts().plot(kind=\"pie\",\n                                           labels=[\"Disaster(43%)\", \"Not a Disaster(57%)\"],\n                                           colors=['lightcoral','lightskyblue'],\n                                           fontsize=14,\n                                           ylabel=\"\");\n\nplt.subplot(1, 2, 2)\nsns.countplot(x=\"target\",data=df_train, palette=\"RdBu\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:27.967173Z","iopub.execute_input":"2022-07-13T17:37:27.967410Z","iopub.status.idle":"2022-07-13T17:37:28.214347Z","shell.execute_reply.started":"2022-07-13T17:37:27.967388Z","shell.execute_reply":"2022-07-13T17:37:28.213438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#506f3f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 4px;color:white;\"><b>2.4 Other Features distribution</b></p>\n</div>\n\n\n### 1. Tweets length","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.subplot(1, 2, 1)\ndf_train.query(\"target==1\")[\"text\"].str.len().plot(kind=\"hist\",\n                                                  color=\"red\",\n                                                  title=\"Disaster tweets\");\n\nplt.subplot(1, 2, 2)\ndf_train.query(\"target==0\")[\"text\"].str.len().plot(kind=\"hist\",\n                                                  color=\"green\",\n                                                  title=\"Non Disaster tweets\");\n\nplt.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:28.660610Z","iopub.execute_input":"2022-07-13T17:37:28.661236Z","iopub.status.idle":"2022-07-13T17:37:28.983511Z","shell.execute_reply.started":"2022-07-13T17:37:28.661200Z","shell.execute_reply":"2022-07-13T17:37:28.982443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Words in a tweet","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.subplot(1, 2, 1)\ndf_train.query(\"target==1\").text.map(lambda x: len(x.split())).plot(kind=\"hist\",\n                                                                    color=\"red\",\n                                                                    title=\"Disaster tweets\");\n\nplt.subplot(1, 2, 2)\ndf_train.query(\"target==0\").text.map(lambda x: len(x.split())).plot(kind=\"hist\",\n                                                                    color=\"green\",\n                                                                    title=\"Disaster tweets\");\nplt.suptitle('Number of Words in each tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:29.540874Z","iopub.execute_input":"2022-07-13T17:37:29.541576Z","iopub.status.idle":"2022-07-13T17:37:29.881689Z","shell.execute_reply.started":"2022-07-13T17:37:29.541510Z","shell.execute_reply":"2022-07-13T17:37:29.880799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Average length of each word in a tweet","metadata":{}},{"cell_type":"code","source":"df_train.query(\"target==1\")[\"text\"].str.split().map(lambda x: [len(i) for i in x])[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:30.112416Z","iopub.execute_input":"2022-07-13T17:37:30.113279Z","iopub.status.idle":"2022-07-13T17:37:30.140338Z","shell.execute_reply.started":"2022-07-13T17:37:30.113235Z","shell.execute_reply":"2022-07-13T17:37:30.139512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.query(\"target==1\")[\"text\"].str.split().map(lambda x: [len(i) for i in x]).map(lambda x: np.mean(x))[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:30.531656Z","iopub.execute_input":"2022-07-13T17:37:30.532313Z","iopub.status.idle":"2022-07-13T17:37:30.605994Z","shell.execute_reply.started":"2022-07-13T17:37:30.532277Z","shell.execute_reply":"2022-07-13T17:37:30.604787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(10,5))\n\neach_words_len = df_train.query(\"target==1\").text.str.split().map(lambda x: [len(i) for i in x])\nsns.distplot(each_words_len.map(lambda x: np.mean(x)), kde=True, ax=axes[0], color=\"red\");\n\neach_words_len = df_train.query(\"target==0\").text.str.split().map(lambda x: [len(i) for i in x])\nsns.distplot(each_words_len.map(lambda x: np.mean(x)), kde=True, ax=axes[1], color='green');\n\nfig.suptitle('Average word length in each tweet');","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:37:31.081971Z","iopub.execute_input":"2022-07-13T17:37:31.082308Z","iopub.status.idle":"2022-07-13T17:37:31.709173Z","shell.execute_reply.started":"2022-07-13T17:37:31.082277Z","shell.execute_reply":"2022-07-13T17:37:31.706951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Punctuation used in tweets","metadata":{}},{"cell_type":"code","source":"import string\nstring.punctuation","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:39:50.994345Z","iopub.execute_input":"2022-07-13T17:39:50.994724Z","iopub.status.idle":"2022-07-13T17:39:51.001767Z","shell.execute_reply.started":"2022-07-13T17:39:50.994691Z","shell.execute_reply":"2022-07-13T17:39:51.000732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\ndef plot_punctuations(df, target):\n    \n    punctations_dict = defaultdict(int)\n    \n    for idx, text in df[df[\"target\"]==target].text.iteritems():\n        for token in text.split():\n            if token in string.punctuation:\n                punctations_dict[token] +=1   \n\n    return dict(sorted(punctations_dict.items(), key=lambda x: x[1], reverse=True))\n\n\ndt_punctuations = plot_punctuations(df_train, target=1)\nndt_punctuations = plot_punctuations(df_train, target=0)\n\nplt.figure(figsize=(20, 5))\nplt.subplot(121)\nx1, y1 = zip(*dt_punctuations.items())\nplt.bar(x1,y1, color=\"red\", label=\"Disaster tweets\")\nplt.legend()\n\nplt.subplot(122)\nx2, y2 = zip(*ndt_punctuations.items())\nplt.bar(x2, y2, color=\"green\", label=\"Non-Disaster tweets\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:39:53.742905Z","iopub.execute_input":"2022-07-13T17:39:53.743332Z","iopub.status.idle":"2022-07-13T17:39:54.656432Z","shell.execute_reply.started":"2022-07-13T17:39:53.743299Z","shell.execute_reply":"2022-07-13T17:39:54.655539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remarks:\n- Most the diaster and non-diaster tweets have moreover same punctuations used.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 5))\ndf_train.location.value_counts().sort_values(ascending=False)[:30].plot(kind=\"bar\",\n                                                                        color='lightcoral',\n                                                                        linewidth=2,\n                                                                        fontsize=14);\nplt.title('Location Count', fontsize = 20)\nplt.axhline(y=20, color=\"green\")\nplt.axhline(y=10, color=\"blue\")\nplt.xlabel('Regions', fontsize = 12)\nplt.ylabel('Count', fontsize = 12)\nplt.xticks(fontsize=12, rotation=70);","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:39:59.035942Z","iopub.execute_input":"2022-07-13T17:39:59.036294Z","iopub.status.idle":"2022-07-13T17:39:59.402109Z","shell.execute_reply.started":"2022-07-13T17:39:59.036265Z","shell.execute_reply":"2022-07-13T17:39:59.401197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Word Clouds","metadata":{}},{"cell_type":"code","source":"stop_words= set(stopwords.words(\"english\"))\n\n# updating the stopwords considering the \nstop_words.update(['https', 'http', 'amp', 'CO', 't', 'u', 'new', \"I'm\", \"would\"])\n\nwc = WordCloud(width=800,\n               height=400,\n               max_words=200,\n               stopwords=stop_words,\n               background_color= \"black\", \n               colormap=\"Paired\",\n               max_font_size=150)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:01.066885Z","iopub.execute_input":"2022-07-13T17:40:01.068823Z","iopub.status.idle":"2022-07-13T17:40:01.075678Z","shell.execute_reply.started":"2022-07-13T17:40:01.068774Z","shell.execute_reply":"2022-07-13T17:40:01.074766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disaster_tweets_text = df_train.query(\"target==1\").text\nconcat_disaster_tweets_text = disaster_tweets_text.str.cat(sep=\" \")  # str.cat -- string concatenation\n\nnon_disaster_tweets_text = df_train.query(\"target==0\").text\nconcat_non_disaster_tweets_text = non_disaster_tweets_text.str.cat(sep=\" \")\n\nprint('\\033[1m'\"\\nWord Cloud for Disaster Tweets\"'\\033[0m')\nwc.generate(concat_disaster_tweets_text)\nplt.figure(figsize=(12, 5))\nplt.imshow(wc, interpolation='bilinear')\nplt.axis('off')\nplt.show()\n\nprint('\\033[1m'\"\\nWord Cloud for Non-Disaster Tweets\"'\\033[0m')\nwc.generate(concat_non_disaster_tweets_text)\nplt.figure(figsize=(12, 5))\nplt.imshow(wc, interpolation='bilinear')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:01.612010Z","iopub.execute_input":"2022-07-13T17:40:01.612640Z","iopub.status.idle":"2022-07-13T17:40:03.989782Z","shell.execute_reply.started":"2022-07-13T17:40:01.612595Z","shell.execute_reply":"2022-07-13T17:40:03.979591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n# Data Cleaning:","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:38:24.371114Z","iopub.execute_input":"2022-07-13T15:38:24.372095Z","iopub.status.idle":"2022-07-13T15:38:24.378943Z","shell.execute_reply.started":"2022-07-13T15:38:24.372058Z","shell.execute_reply":"2022-07-13T15:38:24.377613Z"}}},{"cell_type":"markdown","source":"### Merging the df_train and df_test for preprocessing","metadata":{}},{"cell_type":"code","source":"df_train[\"istrain\"] = True\ndf_test[\"istrain\"] = False","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:03.991754Z","iopub.execute_input":"2022-07-13T17:40:03.992342Z","iopub.status.idle":"2022-07-13T17:40:04.002697Z","shell.execute_reply.started":"2022-07-13T17:40:03.992303Z","shell.execute_reply":"2022-07-13T17:40:04.001818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape[0] +  df_test.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:04.004177Z","iopub.execute_input":"2022-07-13T17:40:04.004743Z","iopub.status.idle":"2022-07-13T17:40:04.014372Z","shell.execute_reply.started":"2022-07-13T17:40:04.004708Z","shell.execute_reply":"2022-07-13T17:40:04.013560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df_train, df_test], axis=0)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:04.016813Z","iopub.execute_input":"2022-07-13T17:40:04.017757Z","iopub.status.idle":"2022-07-13T17:40:04.052346Z","shell.execute_reply.started":"2022-07-13T17:40:04.017716Z","shell.execute_reply":"2022-07-13T17:40:04.051432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. Removing URLs","metadata":{}},{"cell_type":"code","source":"example = df_train.loc[57, \"text\"]\nexample","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:04.056250Z","iopub.execute_input":"2022-07-13T17:40:04.058555Z","iopub.status.idle":"2022-07-13T17:40:04.068147Z","shell.execute_reply.started":"2022-07-13T17:40:04.058494Z","shell.execute_reply":"2022-07-13T17:40:04.067249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URLs(text):\n    return re.sub(r'http\\S+', ' ', text,  flags=re.MULTILINE)\n\nremove_URLs(example)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:04.072723Z","iopub.execute_input":"2022-07-13T17:40:04.073295Z","iopub.status.idle":"2022-07-13T17:40:04.084909Z","shell.execute_reply.started":"2022-07-13T17:40:04.073255Z","shell.execute_reply":"2022-07-13T17:40:04.083918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']= df['text'].apply(lambda x : remove_URLs(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:04.125392Z","iopub.execute_input":"2022-07-13T17:40:04.125726Z","iopub.status.idle":"2022-07-13T17:40:04.180093Z","shell.execute_reply.started":"2022-07-13T17:40:04.125694Z","shell.execute_reply":"2022-07-13T17:40:04.179278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Remove special characters/punctions","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:05.464855Z","iopub.execute_input":"2022-07-13T16:20:05.466488Z","iopub.status.idle":"2022-07-13T16:20:05.473840Z","shell.execute_reply.started":"2022-07-13T16:20:05.466419Z","shell.execute_reply":"2022-07-13T16:20:05.472123Z"}}},{"cell_type":"code","source":"example = \"Set our hearts ablaze and every city was a gift And every skyline was like a kiss upon the lips @\\x89Û_\"","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:05.361737Z","iopub.execute_input":"2022-07-13T17:40:05.362083Z","iopub.status.idle":"2022-07-13T17:40:05.367149Z","shell.execute_reply.started":"2022-07-13T17:40:05.362052Z","shell.execute_reply":"2022-07-13T17:40:05.365957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lower_text_and_remove_special_chars(text):\n    text = text.lower().strip()\n    return re.sub(r\"\\W+\", \" \", text)\n\nlower_text_and_remove_special_chars(example)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:05.802204Z","iopub.execute_input":"2022-07-13T17:40:05.802544Z","iopub.status.idle":"2022-07-13T17:40:05.812964Z","shell.execute_reply.started":"2022-07-13T17:40:05.802497Z","shell.execute_reply":"2022-07-13T17:40:05.811623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']= df['text'].apply(lambda x : lower_text_and_remove_special_chars(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:06.120312Z","iopub.execute_input":"2022-07-13T17:40:06.120983Z","iopub.status.idle":"2022-07-13T17:40:06.223130Z","shell.execute_reply.started":"2022-07-13T17:40:06.120952Z","shell.execute_reply":"2022-07-13T17:40:06.222274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Removing HTML tags","metadata":{}},{"cell_type":"code","source":"example = \"\"\"<div>\n<h1>Real or Fake</h1>\n<p>Kaggle </p>\n<a href=\"https://www.kaggle.com/c/nlp-getting-started\">getting started</a>\n</div>\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:06.436260Z","iopub.execute_input":"2022-07-13T17:40:06.436778Z","iopub.status.idle":"2022-07-13T17:40:06.441396Z","shell.execute_reply.started":"2022-07-13T17:40:06.436747Z","shell.execute_reply":"2022-07-13T17:40:06.440217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bs4 import BeautifulSoup\n\ndef remove_html_tags(text):\n    return BeautifulSoup(text).get_text()\n\nprint(remove_html_tags(example))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:06.758332Z","iopub.execute_input":"2022-07-13T17:40:06.759064Z","iopub.status.idle":"2022-07-13T17:40:06.767066Z","shell.execute_reply.started":"2022-07-13T17:40:06.759014Z","shell.execute_reply":"2022-07-13T17:40:06.765958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']= df['text'].apply(lambda x : remove_html_tags(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:07.072088Z","iopub.execute_input":"2022-07-13T17:40:07.072771Z","iopub.status.idle":"2022-07-13T17:40:09.256790Z","shell.execute_reply.started":"2022-07-13T17:40:07.072732Z","shell.execute_reply":"2022-07-13T17:40:09.255814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Removing Emojis\n","metadata":{}},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\nremove_emoji(\"Omg another Earthquake 😔😔\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.258543Z","iopub.execute_input":"2022-07-13T17:40:09.258860Z","iopub.status.idle":"2022-07-13T17:40:09.266879Z","shell.execute_reply.started":"2022-07-13T17:40:09.258832Z","shell.execute_reply":"2022-07-13T17:40:09.265946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']= df['text'].apply(lambda x : remove_emoji(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.268279Z","iopub.execute_input":"2022-07-13T17:40:09.268871Z","iopub.status.idle":"2022-07-13T17:40:09.346493Z","shell.execute_reply.started":"2022-07-13T17:40:09.268832Z","shell.execute_reply":"2022-07-13T17:40:09.345660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5. decontraction","metadata":{}},{"cell_type":"code","source":"example = \"I'm the king!\"","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.349096Z","iopub.execute_input":"2022-07-13T17:40:09.349435Z","iopub.status.idle":"2022-07-13T17:40:09.353408Z","shell.execute_reply.started":"2022-07-13T17:40:09.349401Z","shell.execute_reply":"2022-07-13T17:40:09.352316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decontraction_text(text):    \n    # performing de-contraction\n    text = text.replace(\",000,000\", \"m\").replace(\",000\", \"k\").replace(\"′\", \"'\").replace(\"’\", \"'\")\\\n                .replace(\"won't\", \"will not\").replace(\"cannot\", \"can not\").replace(\"can't\", \"can not\")\\\n                .replace(\"n't\", \" not\").replace(\"what's\", \"what is\").replace(\"it's\", \"it is\")\\\n                .replace(\"'ve\", \" have\").replace(\"i'm\", \"i am\").replace(\"'re\", \" are\")\\\n                .replace(\"he's\", \"he is\").replace(\"she's\", \"she is\").replace(\"'s\", \"is\")\\\n                .replace(\"'m\", \"am\").replace(\"'t\", \"not\")\\\n                .replace(\"'ll\", \" will\")\n    \n    return text\n\ndecontraction_text(example)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.355047Z","iopub.execute_input":"2022-07-13T17:40:09.355397Z","iopub.status.idle":"2022-07-13T17:40:09.369493Z","shell.execute_reply.started":"2022-07-13T17:40:09.355363Z","shell.execute_reply":"2022-07-13T17:40:09.368507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']= df['text'].apply(lambda x : decontraction_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.370761Z","iopub.execute_input":"2022-07-13T17:40:09.371811Z","iopub.status.idle":"2022-07-13T17:40:09.414902Z","shell.execute_reply.started":"2022-07-13T17:40:09.371774Z","shell.execute_reply":"2022-07-13T17:40:09.414039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spelling Correction","metadata":{}},{"cell_type":"code","source":"!pip install pyspellchecker","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:09.416225Z","iopub.execute_input":"2022-07-13T17:40:09.416576Z","iopub.status.idle":"2022-07-13T17:40:18.458299Z","shell.execute_reply.started":"2022-07-13T17:40:09.416542Z","shell.execute_reply":"2022-07-13T17:40:18.457114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spellchecker import SpellChecker\n\nspell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ntext = \"corect me plese\"\ncorrect_spellings(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.460344Z","iopub.execute_input":"2022-07-13T17:40:18.460747Z","iopub.status.idle":"2022-07-13T17:40:18.608557Z","shell.execute_reply.started":"2022-07-13T17:40:18.460706Z","shell.execute_reply":"2022-07-13T17:40:18.607500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df['text']=df['text'].apply(lambda x : correct_spellings(x)#)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.609908Z","iopub.execute_input":"2022-07-13T17:40:18.610929Z","iopub.status.idle":"2022-07-13T17:40:18.616123Z","shell.execute_reply.started":"2022-07-13T17:40:18.610888Z","shell.execute_reply":"2022-07-13T17:40:18.615184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n# Data Preparation for model","metadata":{}},{"cell_type":"code","source":"train = df.query(\"istrain==True\")\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.620985Z","iopub.execute_input":"2022-07-13T17:40:18.621408Z","iopub.status.idle":"2022-07-13T17:40:18.636463Z","shell.execute_reply.started":"2022-07-13T17:40:18.621377Z","shell.execute_reply":"2022-07-13T17:40:18.635620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = df.query(\"istrain==False\")\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.639323Z","iopub.execute_input":"2022-07-13T17:40:18.639596Z","iopub.status.idle":"2022-07-13T17:40:18.652204Z","shell.execute_reply.started":"2022-07-13T17:40:18.639571Z","shell.execute_reply":"2022-07-13T17:40:18.651339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.653963Z","iopub.execute_input":"2022-07-13T17:40:18.654958Z","iopub.status.idle":"2022-07-13T17:40:18.673751Z","shell.execute_reply.started":"2022-07-13T17:40:18.654921Z","shell.execute_reply":"2022-07-13T17:40:18.672641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initiating the input and labels\nx_train, x_valid, y_train, y_valid = model_selection.train_test_split(train.text,\n                                                                      train.target, \n                                                                      test_size=0.15, \n                                                                      random_state=42, \n                                                                      stratify=train.target) \n\ntrain_sentences = np.array(x_train) \ntrain_labels = np.array(y_train)\n\nvalid_sentences = np.array(x_valid)\nvalid_labels = np.array(y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.675623Z","iopub.execute_input":"2022-07-13T17:40:18.676010Z","iopub.status.idle":"2022-07-13T17:40:18.691359Z","shell.execute_reply.started":"2022-07-13T17:40:18.675976Z","shell.execute_reply":"2022-07-13T17:40:18.690560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building the Model and Word Embedding","metadata":{}},{"cell_type":"code","source":"# url from tf_hub\nimport tensorflow_hub as hub\n\nuniversal_sentence_encoder_embedding = hub.KerasLayer(\"https://tfhub.dev/google/universal-sentence-encoder-large/5\", \n                                                      trainable=False,\n                                                      input_shape=[],\n                                                      dtype=tf.string,\n                                                      name = \"universal-sentence-encoder\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:18.694265Z","iopub.execute_input":"2022-07-13T17:40:18.694517Z","iopub.status.idle":"2022-07-13T17:40:32.398723Z","shell.execute_reply.started":"2022-07-13T17:40:18.694494Z","shell.execute_reply":"2022-07-13T17:40:32.397702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    inputs = keras.Input(shape=(), dtype=tf.string, name='text')\n    x = universal_sentence_encoder_embedding(inputs)\n    x = layers.Dropout(0.5)(x)\n    x = layers.Dense(32, activation=\"relu\")(x)\n    outputs = layers.Dense(1, activation=\"sigmoid\", name=\"output\")(x)\n\n    model = keras.Model(inputs, outputs, name=\"USE_dense_model_1\")\n    \n    model.compile(\n        optimizer= keras.optimizers.Adam(),\n        loss=\"binary_crossentropy\",\n        metrics=[\"accuracy\"]\n    )\n    \n    return model\n\nmodel = get_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:32.400296Z","iopub.execute_input":"2022-07-13T17:40:32.400665Z","iopub.status.idle":"2022-07-13T17:40:34.575396Z","shell.execute_reply.started":"2022-07-13T17:40:32.400628Z","shell.execute_reply":"2022-07-13T17:40:34.574205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs=40\nbatch_size=32\n\ncallbacks = [\n        keras.callbacks.EarlyStopping(patience=2, monitor='val_loss')\n]\n\nhistory = model.fit(x=train_sentences,\n                    y=train_labels,\n                    epochs=num_epochs,\n                    batch_size=batch_size,\n                    validation_data=(valid_sentences, valid_labels),\n                    callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:34.577042Z","iopub.execute_input":"2022-07-13T17:40:34.577422Z","iopub.status.idle":"2022-07-13T17:43:12.302666Z","shell.execute_reply.started":"2022-07-13T17:40:34.577385Z","shell.execute_reply":"2022-07-13T17:43:12.301563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sequences = np.array(test.text)\npreds = model.predict(test_sequences)\npreds_final = tf.squeeze(tf.round(preds))\npreds_final = np.array(preds_final, dtype=\"int32\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:43:12.304836Z","iopub.execute_input":"2022-07-13T17:43:12.305171Z","iopub.status.idle":"2022-07-13T17:43:21.587031Z","shell.execute_reply.started":"2022-07-13T17:43:12.305143Z","shell.execute_reply":"2022-07-13T17:43:21.586020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(preds_final)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:43:21.589805Z","iopub.execute_input":"2022-07-13T17:43:21.590636Z","iopub.status.idle":"2022-07-13T17:43:21.598873Z","shell.execute_reply.started":"2022-07-13T17:43:21.590592Z","shell.execute_reply":"2022-07-13T17:43:21.597905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/nlp-getting-started/sample_submission.csv\")\nsubmission.target = preds_final\n\n\n\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Submission csv file generated...\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:43:21.603594Z","iopub.execute_input":"2022-07-13T17:43:21.603880Z","iopub.status.idle":"2022-07-13T17:43:21.628150Z","shell.execute_reply.started":"2022-07-13T17:43:21.603844Z","shell.execute_reply":"2022-07-13T17:43:21.627193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 9. References:\n\n- [Notebok: Basic EDA,Cleaning and GloVe](https://www.kaggle.com/code/shahules/basic-eda-cleaning-and-glove) by [Shahules](https://www.kaggle.com/shahules)","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}