{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T17:19:10.793380Z","iopub.execute_input":"2022-08-05T17:19:10.793828Z","iopub.status.idle":"2022-08-05T17:19:10.831935Z","shell.execute_reply.started":"2022-08-05T17:19:10.793743Z","shell.execute_reply":"2022-08-05T17:19:10.830747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 1. Import Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport re\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom bs4 import BeautifulSoup\nimport nltk\nfrom nltk.corpus import stopwords\nfrom string import punctuation\nfrom keras.preprocessing import sequence\nfrom tensorflow import keras\nfrom keras import models\nfrom keras import layers\nimport string\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:31.374503Z","iopub.execute_input":"2022-08-06T19:05:31.374908Z","iopub.status.idle":"2022-08-06T19:05:31.389168Z","shell.execute_reply.started":"2022-08-06T19:05:31.374876Z","shell.execute_reply":"2022-08-06T19:05:31.387534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 2. Import Dataset","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ndf_test = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:33.807693Z","iopub.execute_input":"2022-08-06T19:05:33.808312Z","iopub.status.idle":"2022-08-06T19:05:33.858237Z","shell.execute_reply.started":"2022-08-06T19:05:33.808225Z","shell.execute_reply":"2022-08-06T19:05:33.857032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 3. Data Preparation","metadata":{}},{"cell_type":"code","source":"df_train.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:40.240573Z","iopub.execute_input":"2022-08-06T19:05:40.240999Z","iopub.status.idle":"2022-08-06T19:05:40.264026Z","shell.execute_reply.started":"2022-08-06T19:05:40.240968Z","shell.execute_reply":"2022-08-06T19:05:40.262565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In This work, we only need the text of each tweet and its label target. Therefore, we will drop the rest of features:","metadata":{}},{"cell_type":"code","source":"df_train.drop(['id','keyword','location'],axis=1,inplace=True)\ndf_test.drop(['keyword','location'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:43.221284Z","iopub.execute_input":"2022-08-06T19:05:43.221762Z","iopub.status.idle":"2022-08-06T19:05:43.234096Z","shell.execute_reply.started":"2022-08-06T19:05:43.221725Z","shell.execute_reply":"2022-08-06T19:05:43.232243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:46.003331Z","iopub.execute_input":"2022-08-06T19:05:46.004027Z","iopub.status.idle":"2022-08-06T19:05:46.020904Z","shell.execute_reply.started":"2022-08-06T19:05:46.003980Z","shell.execute_reply":"2022-08-06T19:05:46.019641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can create a column length in train_data, which will have length of each text.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,4))\nsns.countplot(data = df_train, x = 'target')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:49.781832Z","iopub.execute_input":"2022-08-06T19:05:49.782262Z","iopub.status.idle":"2022-08-06T19:05:49.966821Z","shell.execute_reply.started":"2022-08-06T19:05:49.782230Z","shell.execute_reply":"2022-08-06T19:05:49.965858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['length'] = df_train['text'].apply(len)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:52.404048Z","iopub.execute_input":"2022-08-06T19:05:52.405669Z","iopub.status.idle":"2022-08-06T19:05:52.427448Z","shell.execute_reply.started":"2022-08-06T19:05:52.405613Z","shell.execute_reply":"2022-08-06T19:05:52.426097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['length'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:55.065524Z","iopub.execute_input":"2022-08-06T19:05:55.065932Z","iopub.status.idle":"2022-08-06T19:05:55.078300Z","shell.execute_reply.started":"2022-08-06T19:05:55.065900Z","shell.execute_reply":"2022-08-06T19:05:55.077198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[df_train['length']==157].text.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:05:57.666267Z","iopub.execute_input":"2022-08-06T19:05:57.666670Z","iopub.status.idle":"2022-08-06T19:05:57.679023Z","shell.execute_reply.started":"2022-08-06T19:05:57.666635Z","shell.execute_reply":"2022-08-06T19:05:57.677921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here the maximum length word is having repeated punctuations. So the important information delivered is very less.","metadata":{}},{"cell_type":"code","source":"#Plotting tweets length\nplt.figure(figsize=(8,4))\nsns.histplot(data = df_train, x ='length',kde=True, bins=30)\nplt.title('Length of tweets')\nplt.xlabel(\"Number of Characters\")\nplt.ylabel(\"Density\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:00.652636Z","iopub.execute_input":"2022-08-06T19:06:00.653104Z","iopub.status.idle":"2022-08-06T19:06:00.977512Z","shell.execute_reply.started":"2022-08-06T19:06:00.653070Z","shell.execute_reply":"2022-08-06T19:06:00.975713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing Punctuations:**","metadata":{}},{"cell_type":"code","source":"# string.punctuation will give the punctuations.\nstring.punctuation","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:04.592070Z","iopub.execute_input":"2022-08-06T19:06:04.592594Z","iopub.status.idle":"2022-08-06T19:06:04.601025Z","shell.execute_reply.started":"2022-08-06T19:06:04.592556Z","shell.execute_reply":"2022-08-06T19:06:04.599768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can create a function 'clean_text' to remove punctuations, and then it can used to clean text column of training data.","metadata":{}},{"cell_type":"code","source":"def clean_text(text):\n    clean_text = [char for char in text if char not in string.punctuation]\n    clean_text = ''.join(clean_text)\n    return clean_text ","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:06.908415Z","iopub.execute_input":"2022-08-06T19:06:06.908984Z","iopub.status.idle":"2022-08-06T19:06:06.915673Z","shell.execute_reply.started":"2022-08-06T19:06:06.908939Z","shell.execute_reply":"2022-08-06T19:06:06.914404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Function 'clean_text', will remove punctuations in a text. Now we apply it to training data and create a column clean_text for the training data, which will have text without puntuations.","metadata":{}},{"cell_type":"code","source":"df_train['clean_text'] = df_train['text'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:09.456197Z","iopub.execute_input":"2022-08-06T19:06:09.456616Z","iopub.status.idle":"2022-08-06T19:06:09.571422Z","shell.execute_reply.started":"2022-08-06T19:06:09.456584Z","shell.execute_reply":"2022-08-06T19:06:09.569982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:11.613071Z","iopub.execute_input":"2022-08-06T19:06:11.614558Z","iopub.status.idle":"2022-08-06T19:06:11.627563Z","shell.execute_reply.started":"2022-08-06T19:06:11.614494Z","shell.execute_reply":"2022-08-06T19:06:11.626121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#without apply 'clean_text function:\ndf_train.loc[1270,'text']","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:14.457404Z","iopub.execute_input":"2022-08-06T19:06:14.457871Z","iopub.status.idle":"2022-08-06T19:06:14.466355Z","shell.execute_reply.started":"2022-08-06T19:06:14.457836Z","shell.execute_reply":"2022-08-06T19:06:14.465284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#after apply 'clean_text function:\ndf_train.loc[1270,'clean_text']","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:16.483689Z","iopub.execute_input":"2022-08-06T19:06:16.484087Z","iopub.status.idle":"2022-08-06T19:06:16.493043Z","shell.execute_reply.started":"2022-08-06T19:06:16.484055Z","shell.execute_reply":"2022-08-06T19:06:16.491351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing Noise:**","metadata":{}},{"cell_type":"markdown","source":"Noise in a text can be considered as anything which does belong to normal human language interaction.\n\nNoise in the text can generally be considered as URL, abbreviations, emojis, message inside HTML tag, etc. Punctuations can also be considered as noise. But here we have already removed punctuations.\n\nThe main reason why abbreviations are included as noise is that some people write thx for thankyou. If abbreviations are not replaced with the original word, 'thx' and 'thankyou' will be considered as two different words.\n","metadata":{}},{"cell_type":"code","source":"abbreviations = {\n    \"$\" : \" dollar \",\n    \"€\" : \" euro \",\n    \"4ao\" : \"for adults only\",\n    \"a.m\" : \"before midday\",\n    \"a3\" : \"anytime anywhere anyplace\",\n    \"aamof\" : \"as a matter of fact\",\n    \"acct\" : \"account\",\n    \"adih\" : \"another day in hell\",\n    \"afaic\" : \"as far as i am concerned\",\n    \"afaict\" : \"as far as i can tell\",\n    \"afaik\" : \"as far as i know\",\n    \"afair\" : \"as far as i remember\",\n    \"afk\" : \"away from keyboard\",\n    \"app\" : \"application\",\n    \"approx\" : \"approximately\",\n    \"apps\" : \"applications\",\n    \"asap\" : \"as soon as possible\",\n    \"asl\" : \"age, sex, location\",\n    \"atk\" : \"at the keyboard\",\n    \"ave.\" : \"avenue\",\n    \"aymm\" : \"are you my mother\",\n    \"ayor\" : \"at your own risk\", \n    \"b&b\" : \"bed and breakfast\",\n    \"b+b\" : \"bed and breakfast\",\n    \"b.c\" : \"before christ\",\n    \"b2b\" : \"business to business\",\n    \"b2c\" : \"business to customer\",\n    \"b4\" : \"before\",\n    \"b4n\" : \"bye for now\",\n    \"b@u\" : \"back at you\",\n    \"bae\" : \"before anyone else\",\n    \"bak\" : \"back at keyboard\",\n    \"bbbg\" : \"bye bye be good\",\n    \"bbc\" : \"british broadcasting corporation\",\n    \"bbias\" : \"be back in a second\",\n    \"bbl\" : \"be back later\",\n    \"bbs\" : \"be back soon\",\n    \"be4\" : \"before\",\n    \"bfn\" : \"bye for now\",\n    \"blvd\" : \"boulevard\",\n    \"bout\" : \"about\",\n    \"brb\" : \"be right back\",\n    \"bros\" : \"brothers\",\n    \"brt\" : \"be right there\",\n    \"bsaaw\" : \"big smile and a wink\",\n    \"btw\" : \"by the way\",\n    \"bwl\" : \"bursting with laughter\",\n    \"c/o\" : \"care of\",\n    \"cet\" : \"central european time\",\n    \"cf\" : \"compare\",\n    \"cia\" : \"central intelligence agency\",\n    \"csl\" : \"can not stop laughing\",\n    \"cu\" : \"see you\",\n    \"cul8r\" : \"see you later\",\n    \"cv\" : \"curriculum vitae\",\n    \"cwot\" : \"complete waste of time\",\n    \"cya\" : \"see you\",\n    \"cyt\" : \"see you tomorrow\",\n    \"dae\" : \"does anyone else\",\n    \"dbmib\" : \"do not bother me i am busy\",\n    \"diy\" : \"do it yourself\",\n    \"dm\" : \"direct message\",\n    \"dwh\" : \"during work hours\",\n    \"e123\" : \"easy as one two three\",\n    \"eet\" : \"eastern european time\",\n    \"eg\" : \"example\",\n    \"embm\" : \"early morning business meeting\",\n    \"encl\" : \"enclosed\",\n    \"encl.\" : \"enclosed\",\n    \"etc\" : \"and so on\",\n    \"faq\" : \"frequently asked questions\",\n    \"fawc\" : \"for anyone who cares\",\n    \"fb\" : \"facebook\",\n    \"fc\" : \"fingers crossed\",\n    \"fig\" : \"figure\",\n    \"fimh\" : \"forever in my heart\", \n    \"ft.\" : \"feet\",\n    \"ft\" : \"featuring\",\n    \"ftl\" : \"for the loss\",\n    \"ftw\" : \"for the win\",\n    \"fwiw\" : \"for what it is worth\",\n    \"fyi\" : \"for your information\",\n    \"g9\" : \"genius\",\n    \"gahoy\" : \"get a hold of yourself\",\n    \"gal\" : \"get a life\",\n    \"gcse\" : \"general certificate of secondary education\",\n    \"gfn\" : \"gone for now\",\n    \"gg\" : \"good game\",\n    \"gl\" : \"good luck\",\n    \"glhf\" : \"good luck have fun\",\n    \"gmt\" : \"greenwich mean time\",\n    \"gmta\" : \"great minds think alike\",\n    \"gn\" : \"good night\",\n    \"g.o.a.t\" : \"greatest of all time\",\n    \"goat\" : \"greatest of all time\",\n    \"goi\" : \"get over it\",\n    \"gps\" : \"global positioning system\",\n    \"gr8\" : \"great\",\n    \"gratz\" : \"congratulations\",\n    \"gyal\" : \"girl\",\n    \"h&c\" : \"hot and cold\",\n    \"hp\" : \"horsepower\",\n    \"hr\" : \"hour\",\n    \"hrh\" : \"his royal highness\",\n    \"ht\" : \"height\",\n    \"ibrb\" : \"i will be right back\",\n    \"ic\" : \"i see\",\n    \"icq\" : \"i seek you\",\n    \"icymi\" : \"in case you missed it\",\n    \"idc\" : \"i do not care\",\n    \"idgadf\" : \"i do not give a damn fuck\",\n    \"idgaf\" : \"i do not give a fuck\",\n    \"idk\" : \"i do not know\",\n    \"ie\" : \"that is\",\n    \"i.e\" : \"that is\",\n    \"ifyp\" : \"i feel your pain\",\n    \"IG\" : \"instagram\",\n    \"iirc\" : \"if i remember correctly\",\n    \"ilu\" : \"i love you\",\n    \"ily\" : \"i love you\",\n    \"imho\" : \"in my humble opinion\",\n    \"imo\" : \"in my opinion\",\n    \"imu\" : \"i miss you\",\n    \"iow\" : \"in other words\",\n    \"irl\" : \"in real life\",\n    \"j4f\" : \"just for fun\",\n    \"jic\" : \"just in case\",\n    \"jk\" : \"just kidding\",\n    \"jsyk\" : \"just so you know\",\n    \"l8r\" : \"later\",\n    \"lb\" : \"pound\",\n    \"lbs\" : \"pounds\",\n    \"ldr\" : \"long distance relationship\",\n    \"lmao\" : \"laugh my ass off\",\n    \"lmfao\" : \"laugh my fucking ass off\",\n    \"lol\" : \"laughing out loud\",\n    \"ltd\" : \"limited\",\n    \"ltns\" : \"long time no see\",\n    \"m8\" : \"mate\",\n    \"mf\" : \"motherfucker\",\n    \"mfs\" : \"motherfuckers\",\n    \"mfw\" : \"my face when\",\n    \"mofo\" : \"motherfucker\",\n    \"mph\" : \"miles per hour\",\n    \"mr\" : \"mister\",\n    \"mrw\" : \"my reaction when\",\n    \"ms\" : \"miss\",\n    \"mte\" : \"my thoughts exactly\",\n    \"nagi\" : \"not a good idea\",\n    \"nbc\" : \"national broadcasting company\",\n    \"nbd\" : \"not big deal\",\n    \"nfs\" : \"not for sale\",\n    \"ngl\" : \"not going to lie\",\n    \"nhs\" : \"national health service\",\n    \"nrn\" : \"no reply necessary\",\n    \"nsfl\" : \"not safe for life\",\n    \"nsfw\" : \"not safe for work\",\n    \"nth\" : \"nice to have\",\n    \"nvr\" : \"never\",\n    \"nyc\" : \"new york city\",\n    \"oc\" : \"original content\",\n    \"og\" : \"original\",\n    \"ohp\" : \"overhead projector\",\n    \"oic\" : \"oh i see\",\n    \"omdb\" : \"over my dead body\",\n    \"omg\" : \"oh my god\",\n    \"omw\" : \"on my way\",\n    \"p.a\" : \"per annum\",\n    \"p.m\" : \"after midday\",\n    \"pm\" : \"prime minister\",\n    \"poc\" : \"people of color\",\n    \"pov\" : \"point of view\",\n    \"pp\" : \"pages\",\n    \"ppl\" : \"people\",\n    \"prw\" : \"parents are watching\",\n    \"ps\" : \"postscript\",\n    \"pt\" : \"point\",\n    \"ptb\" : \"please text back\",\n    \"pto\" : \"please turn over\",\n    \"qpsa\" : \"what happens\",\n    \"ratchet\" : \"rude\",\n    \"rbtl\" : \"read between the lines\",\n    \"rlrt\" : \"real life retweet\", \n    \"rofl\" : \"rolling on the floor laughing\",\n    \"roflol\" : \"rolling on the floor laughing out loud\",\n    \"rotflmao\" : \"rolling on the floor laughing my ass off\",\n    \"rt\" : \"retweet\",\n    \"ruok\" : \"are you ok\",\n    \"sfw\" : \"safe for work\",\n    \"sk8\" : \"skate\",\n    \"smh\" : \"shake my head\",\n    \"sq\" : \"square\",\n    \"srsly\" : \"seriously\", \n    \"ssdd\" : \"same stuff different day\",\n    \"tbh\" : \"to be honest\",\n    \"tbs\" : \"tablespooful\",\n    \"tbsp\" : \"tablespooful\",\n    \"tfw\" : \"that feeling when\",\n    \"thks\" : \"thank you\",\n    \"tho\" : \"though\",\n    \"thx\" : \"thank you\",\n    \"tia\" : \"thanks in advance\",\n    \"til\" : \"today i learned\",\n    \"tl;dr\" : \"too long i did not read\",\n    \"tldr\" : \"too long i did not read\",\n    \"tmb\" : \"tweet me back\",\n    \"tntl\" : \"trying not to laugh\",\n    \"ttyl\" : \"talk to you later\",\n    \"u\" : \"you\",\n    \"u2\" : \"you too\",\n    \"u4e\" : \"yours for ever\",\n    \"utc\" : \"coordinated universal time\",\n    \"w/\" : \"with\",\n    \"w/o\" : \"without\",\n    \"w8\" : \"wait\",\n    \"wassup\" : \"what is up\",\n    \"wb\" : \"welcome back\",\n    \"wtf\" : \"what the fuck\",\n    \"wtg\" : \"way to go\",\n    \"wtpa\" : \"where the party at\",\n    \"wuf\" : \"where are you from\",\n    \"wuzup\" : \"what is up\",\n    \"wywh\" : \"wish you were here\",\n    \"yd\" : \"yard\",\n    \"ygtr\" : \"you got that right\",\n    \"ynk\" : \"you never know\",\n    \"zzz\" : \"sleeping bored and tired\"\n}","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2022-08-06T19:06:20.017512Z","iopub.execute_input":"2022-08-06T19:06:20.018866Z","iopub.status.idle":"2022-08-06T19:06:20.047824Z","shell.execute_reply.started":"2022-08-06T19:06:20.018810Z","shell.execute_reply":"2022-08-06T19:06:20.046100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove all URLs, replace by URL\ndef remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'URL',text)\n\n# Remove HTML beacon\ndef remove_HTML(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)\n\n# Remove non printable characters\ndef remove_not_ASCII(text):\n    text = ''.join([word for word in text if word in string.printable])\n    return text\n\n# Change an abbreviation by its true meaning\ndef word_abbrev(word):\n    return abbreviations[word.lower()] if word.lower() in abbreviations.keys() else word\n\n# Replace all abbreviations\ndef replace_abbrev(text):\n    string = \"\"\n    for word in text.split():\n        string += word_abbrev(word) + \" \"        \n    return string\n\n# Remove @ and mention, replace by USER\ndef remove_mention(text):\n    at=re.compile(r'@\\S+')\n    return at.sub(r'USER',text)\n\n# Remove numbers, replace it by NUMBER\ndef remove_number(text):\n    num = re.compile(r'[-+]?[.\\d]*[\\d]+[:,.\\d]*')\n    return num.sub(r'NUMBER', text)\n\n# Remove all emojis, replace by EMOJI\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'EMOJI', text)\n\n# Replace some others smileys with SADFACE\ndef transcription_sad(text):\n    eyes = \"[8:=;]\"\n    nose = \"['`\\-]\"\n    smiley = re.compile(r'[8:=;][\\'\\-]?[(\\\\/]')\n    return smiley.sub(r'SADFACE', text)\n\n# Replace some smileys with SMILE\ndef transcription_smile(text):\n    eyes = \"[8:=;]\"\n    nose = \"['`\\-]\"\n    smiley = re.compile(r'[8:=;][\\'\\-]?[)dDp]')\n    #smiley = re.compile(r'#{eyes}#{nose}[)d]+|[)d]+#{nose}#{eyes}/i')\n    return smiley.sub(r'SMILE', text)\n\n# Replace <3 with HEART\ndef transcription_heart(text):\n    heart = re.compile(r'<3')\n    return heart.sub(r'HEART', text)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:25.698696Z","iopub.execute_input":"2022-08-06T19:06:25.699154Z","iopub.status.idle":"2022-08-06T19:06:25.713285Z","shell.execute_reply.started":"2022-08-06T19:06:25.699096Z","shell.execute_reply":"2022-08-06T19:06:25.711699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_tweet(text):\n    \n    # Remove non text\n    text = remove_URL(text)\n    text = remove_HTML(text)\n    text = remove_not_ASCII(text)\n    \n    # replace abbreviations, @ and number\n    text = replace_abbrev(text)  \n    text = remove_mention(text)\n    text = remove_number(text)\n    \n    # Remove emojis / smileys\n    text = remove_emoji(text)\n    text = transcription_sad(text)\n    text = transcription_smile(text)\n    text = transcription_heart(text)\n  \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:29.574856Z","iopub.execute_input":"2022-08-06T19:06:29.575374Z","iopub.status.idle":"2022-08-06T19:06:29.584403Z","shell.execute_reply.started":"2022-08-06T19:06:29.575335Z","shell.execute_reply":"2022-08-06T19:06:29.582745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Function clean_tweet() will remove all the noise in the text.","metadata":{}},{"cell_type":"code","source":"df_train[\"clean_text\"] = df_train[\"clean_text\"].apply(clean_tweet)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:33.405956Z","iopub.execute_input":"2022-08-06T19:06:33.406469Z","iopub.status.idle":"2022-08-06T19:06:33.766824Z","shell.execute_reply.started":"2022-08-06T19:06:33.406433Z","shell.execute_reply":"2022-08-06T19:06:33.765115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:35.325116Z","iopub.execute_input":"2022-08-06T19:06:35.325615Z","iopub.status.idle":"2022-08-06T19:06:35.339134Z","shell.execute_reply.started":"2022-08-06T19:06:35.325580Z","shell.execute_reply":"2022-08-06T19:06:35.337859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing Stopwords:**","metadata":{}},{"cell_type":"markdown","source":"Stopwords are commonly used words, which do not have any distinguishing features, like \"a\", \"an\", \"the\", so on… and search engine is programmed to ignore them while indexing entries and while retrieving the results of a search query. It saves space in the database and decreases processing speed. \n\nNatural Language Toolkit(nlkt) in python has a list of stopwords stored in 16 different languages. It is a leading platform for building a python program to work with human language data.\n","metadata":{}},{"cell_type":"code","source":"print(stopwords.words('english'))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:40.447808Z","iopub.execute_input":"2022-08-06T19:06:40.448329Z","iopub.status.idle":"2022-08-06T19:06:40.456391Z","shell.execute_reply.started":"2022-08-06T19:06:40.448290Z","shell.execute_reply":"2022-08-06T19:06:40.455183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(text):\n    remove_stopword = [word for word in text.split() if word.lower() not in stopwords.words('english')]\n    return remove_stopword","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:42.973426Z","iopub.execute_input":"2022-08-06T19:06:42.973940Z","iopub.status.idle":"2022-08-06T19:06:42.981751Z","shell.execute_reply.started":"2022-08-06T19:06:42.973901Z","shell.execute_reply":"2022-08-06T19:06:42.980350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"clean_text\"] = df_train[\"clean_text\"].apply(remove_stopwords)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:06:45.196168Z","iopub.execute_input":"2022-08-06T19:06:45.197123Z","iopub.status.idle":"2022-08-06T19:07:00.471133Z","shell.execute_reply.started":"2022-08-06T19:06:45.197075Z","shell.execute_reply":"2022-08-06T19:07:00.469767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:02.542565Z","iopub.execute_input":"2022-08-06T19:07:02.542973Z","iopub.status.idle":"2022-08-06T19:07:02.559938Z","shell.execute_reply.started":"2022-08-06T19:07:02.542942Z","shell.execute_reply":"2022-08-06T19:07:02.558214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lowercasing:**","metadata":{}},{"cell_type":"markdown","source":"Lowercasing is a preprocessing method in which the text is converted into the lower case. In tokenization, Keras tokenizer is used, which will be converting texts to lowercase.\n\nSo here there is no need for lowercasing the texts as it will be a duplicate work.\n","metadata":{}},{"cell_type":"markdown","source":"**Tokenization:**","metadata":{}},{"cell_type":"markdown","source":"Tokenization generally decomposes text documents into small tokens and constructs a document word matrix. A document can be considered as a bag of words. Collection of document is called Corpus.\n\nIn document word matrix :\n\n* Each Row represents a document (bag of words)\n* Each column distinct token\n* Each cell represents the frequency of occurrence of the token\n\nHere Keras Tokenizer() is used which is supported by Tensorflow as a high-level API that encodes the token to a numerical value. The main reason to use this is in LSTM input is provided by embedding layer, which requires input data to be integer encoded.\n\nParameter 'num_words' can be used to restrict the number of the token to considered by the model.\n\nTokenizer() uses fit_on_texts() and texts_to_sequences() to encode the texts to numerical values.\n\n* Fit_on_texts() Updates internal vocabulary based on a list of texts. It will create a dictionary with word mapping with an index (unique numerical value). Here all the words will be in lower case and the least value of index will be the more frequent word.\n\n* texts_to_sequences() Transforms each text in texts to a sequence of numerical value. It will give assign the index of each to the word. So the output will be series of numerical values.\n\n","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:07.236215Z","iopub.execute_input":"2022-08-06T19:07:07.236673Z","iopub.status.idle":"2022-08-06T19:07:07.252898Z","shell.execute_reply.started":"2022-08-06T19:07:07.236639Z","shell.execute_reply":"2022-08-06T19:07:07.251908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = 23000\ntokenizer = Tokenizer(num_words = vocab_size, split='')\ntokenizer.fit_on_texts(df_train['clean_text'].values)\nX = tokenizer.texts_to_sequences(df_train['clean_text'].values)\nX = pad_sequences(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:13.482969Z","iopub.execute_input":"2022-08-06T19:07:13.483567Z","iopub.status.idle":"2022-08-06T19:07:13.685424Z","shell.execute_reply.started":"2022-08-06T19:07:13.483524Z","shell.execute_reply":"2022-08-06T19:07:13.684094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are not be having the same length for all the sentences and while providing input to neural networks, we should have the same dimension for all inputs. So pad_sequence() is used to pad the input so that all the inputs have the same dimension. It will add zeros to the input, in the beginning, to make sure all the inputs have the same dimension.","metadata":{}},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:16.764894Z","iopub.execute_input":"2022-08-06T19:07:16.765351Z","iopub.status.idle":"2022-08-06T19:07:16.774863Z","shell.execute_reply.started":"2022-08-06T19:07:16.765315Z","shell.execute_reply":"2022-08-06T19:07:16.773349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here size of tokenized vector is 20, it is the maximum length of clean_text considering excluding those tokens which does not belong to top 3000 tokens. That is if the maximum length of clean_text is 35, then the 15 token will not be qualified to come under top 3000 tokens.\n\nWe can restrict on enhance dimension of tokenized vector by providing a parameter maxlen to pad_sequence().\n","metadata":{}},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 4. Modeling Using LSTM","metadata":{}},{"cell_type":"code","source":"y = df_train.target","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:25.501598Z","iopub.execute_input":"2022-08-06T19:07:25.502080Z","iopub.status.idle":"2022-08-06T19:07:25.508108Z","shell.execute_reply.started":"2022-08-06T19:07:25.502035Z","shell.execute_reply":"2022-08-06T19:07:25.506998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test = train_test_split(X,y,test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:28.562577Z","iopub.execute_input":"2022-08-06T19:07:28.563009Z","iopub.status.idle":"2022-08-06T19:07:28.571336Z","shell.execute_reply.started":"2022-08-06T19:07:28.562965Z","shell.execute_reply":"2022-08-06T19:07:28.570380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Embedding Layer:**","metadata":{}},{"cell_type":"markdown","source":"\n\nEmbedding layer is the first layer of neural network and it has 3 parameters:\n\n* input_dim: Number of distinct token vector, here it will be 3000 (max_features)\n* output_dim: Dimension of embedding vector, we can take 32 dimension\n* input_length: Size of input layer\n\nHere embedding layer of size will be (3000, 32).\n","metadata":{}},{"cell_type":"code","source":"embed_dim = 32\nlstm_out = 32\nmodel = models.Sequential()\nmodel.add(layers.Embedding(vocab_size,embed_dim,input_length=X.shape[1]))\nmodel.add(layers.Dropout(0.2))\nmodel.add(layers.LSTM(units = lstm_out, dropout=0.2, recurrent_dropout=0.4))\nmodel.add(layers.Dense(1, activation='sigmoid'))\nmodel.compile(optimizer='adam', loss = 'binary_crossentropy', metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:36.522559Z","iopub.execute_input":"2022-08-06T19:07:36.522957Z","iopub.status.idle":"2022-08-06T19:07:36.685749Z","shell.execute_reply.started":"2022-08-06T19:07:36.522926Z","shell.execute_reply":"2022-08-06T19:07:36.684512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:39.342428Z","iopub.execute_input":"2022-08-06T19:07:39.342892Z","iopub.status.idle":"2022-08-06T19:07:39.350872Z","shell.execute_reply.started":"2022-08-06T19:07:39.342858Z","shell.execute_reply":"2022-08-06T19:07:39.349224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Activation Function of Dense layer:**, i.e, is output layer is taken as Sigmoid function is taken as it is good at binary classification and our target column have value either 0 or 1.\n\n**Dropout :** It is added to avoid overfitting.\n\n**Loss Function :** The cross-entropy loss function is an optimization function that is used in the case of training a classification model and binary_crossentropy function computes the cross-entropy loss between true labels and predicted labels.\n\n**optimizer :** Adam is used as optimizer which is replacement optimization algorithm for stochastic gradient descent for training deep learning models. Default learning rate of Adam is 0.001, but here I have initialized it to 0.002.\n\n","metadata":{}},{"cell_type":"code","source":"history = model.fit(X_train,y_train, epochs=15, batch_size=32, validation_data=(X_test,y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:07:44.150786Z","iopub.execute_input":"2022-08-06T19:07:44.151296Z","iopub.status.idle":"2022-08-06T19:10:31.819853Z","shell.execute_reply.started":"2022-08-06T19:07:44.151260Z","shell.execute_reply":"2022-08-06T19:10:31.818271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 5. Evaluation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, recall_score, precision_score, confusion_matrix, f1_score","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:21:15.750533Z","iopub.execute_input":"2022-08-06T19:21:15.751103Z","iopub.status.idle":"2022-08-06T19:21:15.758110Z","shell.execute_reply.started":"2022-08-06T19:21:15.751063Z","shell.execute_reply":"2022-08-06T19:21:15.756525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy of the model on Training Data is - \" , model.evaluate(X_train,y_train)[1]*100 , \"%\")\nprint(\"Accuracy of the model on Testing Data is - \" , model.evaluate(X_test,y_test)[1]*100 , \"%\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:12:13.150880Z","iopub.execute_input":"2022-08-06T19:12:13.151468Z","iopub.status.idle":"2022-08-06T19:12:16.372668Z","shell.execute_reply.started":"2022-08-06T19:12:13.151426Z","shell.execute_reply":"2022-08-06T19:12:16.371595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test).round()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:15:31.126777Z","iopub.execute_input":"2022-08-06T19:15:31.127229Z","iopub.status.idle":"2022-08-06T19:15:31.744812Z","shell.execute_reply.started":"2022-08-06T19:15:31.127194Z","shell.execute_reply":"2022-08-06T19:15:31.743434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def measure(y_true, y_pred):\n    accuracy = round(accuracy_score(y_true, y_pred),4)\n    recall = round(recall_score(y_true, y_pred),4)\n    precision = round(precision_score(y_true, y_pred),4)\n    f1 = round(f1_score(y_true, y_pred),4)\n    return pd.Series({'accuracy_score':accuracy,\n                     'recall_score':recall,\n                     'precision_score':precision,\n                     'f1_score':f1})","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:26:44.273214Z","iopub.execute_input":"2022-08-06T19:26:44.273698Z","iopub.status.idle":"2022-08-06T19:26:44.281922Z","shell.execute_reply.started":"2022-08-06T19:26:44.273665Z","shell.execute_reply":"2022-08-06T19:26:44.280254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"measure(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:27:38.688736Z","iopub.execute_input":"2022-08-06T19:27:38.689325Z","iopub.status.idle":"2022-08-06T19:27:38.710380Z","shell.execute_reply.started":"2022-08-06T19:27:38.689279Z","shell.execute_reply":"2022-08-06T19:27:38.709230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = confusion_matrix(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:28:46.364922Z","iopub.execute_input":"2022-08-06T19:28:46.365421Z","iopub.status.idle":"2022-08-06T19:28:46.374401Z","shell.execute_reply.started":"2022-08-06T19:28:46.365383Z","shell.execute_reply":"2022-08-06T19:28:46.372532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(7, 5))\nsns.heatmap(con, annot=True, fmt='d', cmap='cool')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:28:47.912552Z","iopub.execute_input":"2022-08-06T19:28:47.913072Z","iopub.status.idle":"2022-08-06T19:28:48.099948Z","shell.execute_reply.started":"2022-08-06T19:28:47.913033Z","shell.execute_reply":"2022-08-06T19:28:48.098511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As Accuracy, Precision, recall and F1 score are above 70 % this model can be considered as good model.**\n\n**By changing dimension on Embedding Layer, vocab_size, LSTM can check if the evalution factors are improving or not.**\n","metadata":{}},{"cell_type":"markdown","source":"<a id=\"1\"></a> <br>\n# 6. Submission","metadata":{}},{"cell_type":"markdown","source":"For submission stopwords are not removing, as words like 'not' has a major role in distinguishing disaster and non-disaster tweet.","metadata":{}},{"cell_type":"code","source":"df_test['clean_text'] = df_test['text'].apply(clean_text)\n\ndf_test[\"clean_text\"] = df_test[\"clean_text\"].apply(clean_tweet)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:35:32.409065Z","iopub.execute_input":"2022-08-06T19:35:32.410550Z","iopub.status.idle":"2022-08-06T19:35:32.615238Z","shell.execute_reply.started":"2022-08-06T19:35:32.410494Z","shell.execute_reply":"2022-08-06T19:35:32.614330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['clean_text'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:35:53.850448Z","iopub.execute_input":"2022-08-06T19:35:53.850943Z","iopub.status.idle":"2022-08-06T19:35:53.862458Z","shell.execute_reply.started":"2022-08-06T19:35:53.850897Z","shell.execute_reply":"2022-08-06T19:35:53.860986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = 23000\ntokenizer = Tokenizer(num_words = vocab_size)\ntokenizer.fit_on_texts(df_test['clean_text'].values)\nX = tokenizer.texts_to_sequences(df_test['clean_text'].values)\nX = pad_sequences(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:38:17.153503Z","iopub.execute_input":"2022-08-06T19:38:17.153934Z","iopub.status.idle":"2022-08-06T19:38:17.323041Z","shell.execute_reply.started":"2022-08-06T19:38:17.153899Z","shell.execute_reply":"2022-08-06T19:38:17.321511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = model.predict(X).round()\nsubmission = pd.read_csv(\"/kaggle/input/nlp-getting-started/sample_submission.csv\")\nsubmission['target'] = np.round(y_hat).astype('int')\nsubmission.to_csv('submission.csv', index=False)\nsubmission.describe().style.background_gradient(cmap='coolwarm')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T19:42:28.334097Z","iopub.execute_input":"2022-08-06T19:42:28.335447Z","iopub.status.idle":"2022-08-06T19:42:30.095169Z","shell.execute_reply.started":"2022-08-06T19:42:28.335385Z","shell.execute_reply":"2022-08-06T19:42:30.094194Z"},"trusted":true},"execution_count":null,"outputs":[]}]}