{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Basic Intro \n\nIn this competition, you’re challenged to build a machine learning model that predicts which Tweets are about real disasters and which one’s aren’t.\n\n![](https://encrypted-tbn0.gstatic.com/images?q=tbn%3AANd9GcTigQWzoYCNiDyrz1BN4WTf2X2k9OZ_yvW-FsmcIMsdS9fppNmh)","metadata":{}},{"cell_type":"markdown","source":"## What's in this kernel?\n- Basic EDA\n- Data Cleaning\n- Baseline Model","metadata":{}},{"cell_type":"markdown","source":"### Importing required Libraries.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom keras.optimizers import Adam\nimport warnings\nwarnings.filterwarnings('ignore')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:34.641161Z","iopub.execute_input":"2022-07-10T22:16:34.641462Z","iopub.status.idle":"2022-07-10T22:16:34.652936Z","shell.execute_reply.started":"2022-07-10T22:16:34.641409Z","shell.execute_reply":"2022-07-10T22:16:34.652111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n#os.listdir('../input/glove-global-vectors-for-word-representation/glove.6B.100d.txt')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:34.808337Z","iopub.execute_input":"2022-07-10T22:16:34.808650Z","iopub.status.idle":"2022-07-10T22:16:34.812920Z","shell.execute_reply.started":"2022-07-10T22:16:34.808579Z","shell.execute_reply":"2022-07-10T22:16:34.812132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the data and getting basic idea ","metadata":{}},{"cell_type":"code","source":"train= pd.read_csv('../input/nlp-getting-started/train.csv')\ntest=pd.read_csv('../input/nlp-getting-started/test.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:34.982241Z","iopub.execute_input":"2022-07-10T22:16:34.982538Z","iopub.status.idle":"2022-07-10T22:16:35.024047Z","shell.execute_reply.started":"2022-07-10T22:16:34.982481Z","shell.execute_reply":"2022-07-10T22:16:35.023252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('There are {} rows and {} columns in train'.format(train.shape[0],train.shape[1]))\nprint('There are {} rows and {} columns in test'.format(test.shape[0],test.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:35.131190Z","iopub.execute_input":"2022-07-10T22:16:35.131484Z","iopub.status.idle":"2022-07-10T22:16:35.137605Z","shell.execute_reply.started":"2022-07-10T22:16:35.131432Z","shell.execute_reply":"2022-07-10T22:16:35.136808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class distribution","metadata":{}},{"cell_type":"markdown","source":"Before we begin with anything else,let's check the class distribution.There are only two classes 0 and 1.","metadata":{}},{"cell_type":"code","source":"x=train.target.value_counts()\nsns.barplot(x.index,x)\nplt.gca().set_ylabel('samples')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:35.288097Z","iopub.execute_input":"2022-07-10T22:16:35.288399Z","iopub.status.idle":"2022-07-10T22:16:35.468737Z","shell.execute_reply.started":"2022-07-10T22:16:35.288347Z","shell.execute_reply":"2022-07-10T22:16:35.467876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ohh,as expected ! There is a class distribution.There are more tweets with class 0 ( No disaster) than class 1 ( disaster tweets)","metadata":{}},{"cell_type":"markdown","source":"## Exploratory Data Analysis of tweets","metadata":{}},{"cell_type":"markdown","source":"First,we will do very basic analysis,that is character level,word level and sentence level analysis.","metadata":{}},{"cell_type":"markdown","source":"### Number of characters in tweets","metadata":{}},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=train[train['target']==1]['text'].str.len()\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=train[train['target']==0]['text'].str.len()\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:35.472490Z","iopub.execute_input":"2022-07-10T22:16:35.475841Z","iopub.status.idle":"2022-07-10T22:16:35.919370Z","shell.execute_reply.started":"2022-07-10T22:16:35.475776Z","shell.execute_reply":"2022-07-10T22:16:35.918500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of both seems to be almost same.120 to 140 characters in a tweet are the most common among both.","metadata":{}},{"cell_type":"markdown","source":"### Number of words in a tweet","metadata":{}},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=train[train['target']==1]['text'].str.split().map(lambda x: len(x))\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=train[train['target']==0]['text'].str.split().map(lambda x: len(x))\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Words in a tweet')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:35.921181Z","iopub.execute_input":"2022-07-10T22:16:35.921596Z","iopub.status.idle":"2022-07-10T22:16:36.389128Z","shell.execute_reply.started":"2022-07-10T22:16:35.921545Z","shell.execute_reply":"2022-07-10T22:16:36.388227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  Average word length in a tweet","metadata":{}},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nword=train[train['target']==1]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax1,color='red')\nax1.set_title('disaster')\nword=train[train['target']==0]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax2,color='green')\nax2.set_title('Not disaster')\nfig.suptitle('Average word length in each tweet')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:36.391077Z","iopub.execute_input":"2022-07-10T22:16:36.391580Z","iopub.status.idle":"2022-07-10T22:16:37.315099Z","shell.execute_reply.started":"2022-07-10T22:16:36.391520Z","shell.execute_reply":"2022-07-10T22:16:37.314151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(target):\n    corpus=[]\n    \n    for x in train[train['target']==target]['text'].str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:37.316813Z","iopub.execute_input":"2022-07-10T22:16:37.317339Z","iopub.status.idle":"2022-07-10T22:16:37.323981Z","shell.execute_reply.started":"2022-07-10T22:16:37.317078Z","shell.execute_reply":"2022-07-10T22:16:37.322839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Common stopwords in tweets","metadata":{}},{"cell_type":"markdown","source":"First we  will analyze tweets with class 0.","metadata":{}},{"cell_type":"code","source":"corpus=create_corpus(0)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n        \ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:37.325357Z","iopub.execute_input":"2022-07-10T22:16:37.325781Z","iopub.status.idle":"2022-07-10T22:16:37.377283Z","shell.execute_reply.started":"2022-07-10T22:16:37.325598Z","shell.execute_reply":"2022-07-10T22:16:37.376720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:37.378461Z","iopub.execute_input":"2022-07-10T22:16:37.378810Z","iopub.status.idle":"2022-07-10T22:16:37.786637Z","shell.execute_reply.started":"2022-07-10T22:16:37.378715Z","shell.execute_reply":"2022-07-10T22:16:37.785667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now,we will analyze tweets with class 1.","metadata":{}},{"cell_type":"code","source":"corpus=create_corpus(1)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n    \n\n\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:37.790907Z","iopub.execute_input":"2022-07-10T22:16:37.791166Z","iopub.status.idle":"2022-07-10T22:16:38.201285Z","shell.execute_reply.started":"2022-07-10T22:16:37.791114Z","shell.execute_reply":"2022-07-10T22:16:38.200243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In both of them,\"the\" dominates which is followed by \"a\" in class 0 and \"in\" in class 1.","metadata":{}},{"cell_type":"markdown","source":"### Analyzing punctuations.","metadata":{}},{"cell_type":"markdown","source":"First let's check tweets indicating real disaster.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ncorpus=create_corpus(1)\n\ndic=defaultdict(int)\nimport string\nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i]+=1\n        \nx,y=zip(*dic.items())\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:38.204905Z","iopub.execute_input":"2022-07-10T22:16:38.205476Z","iopub.status.idle":"2022-07-10T22:16:38.652936Z","shell.execute_reply.started":"2022-07-10T22:16:38.205182Z","shell.execute_reply":"2022-07-10T22:16:38.652025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now,we will move on to class 0.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ncorpus=create_corpus(0)\n\ndic=defaultdict(int)\nimport string\nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i]+=1\n        \nx,y=zip(*dic.items())\nplt.bar(x,y,color='green')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:38.658155Z","iopub.execute_input":"2022-07-10T22:16:38.660787Z","iopub.status.idle":"2022-07-10T22:16:39.087801Z","shell.execute_reply.started":"2022-07-10T22:16:38.660705Z","shell.execute_reply":"2022-07-10T22:16:39.086776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Common words ?","metadata":{}},{"cell_type":"code","source":"\ncounter=Counter(corpus)\nmost=counter.most_common()\nx=[]\ny=[]\nfor word,count in most[:40]:\n    if (word not in stop) :\n        x.append(word)\n        y.append(count)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:39.089377Z","iopub.execute_input":"2022-07-10T22:16:39.089678Z","iopub.status.idle":"2022-07-10T22:16:39.125047Z","shell.execute_reply.started":"2022-07-10T22:16:39.089629Z","shell.execute_reply":"2022-07-10T22:16:39.124101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:39.126858Z","iopub.execute_input":"2022-07-10T22:16:39.127453Z","iopub.status.idle":"2022-07-10T22:16:39.370348Z","shell.execute_reply.started":"2022-07-10T22:16:39.127114Z","shell.execute_reply":"2022-07-10T22:16:39.369372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lot of cleaning needed !","metadata":{}},{"cell_type":"markdown","source":"### Ngram analysis","metadata":{}},{"cell_type":"markdown","source":"we will do a bigram (n=2) analysis over the tweets.Let's check the most common bigrams in tweets.","metadata":{}},{"cell_type":"code","source":"def get_top_tweet_bigrams(corpus, n=None):\n    vec = CountVectorizer(ngram_range=(2, 2)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:n]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:39.371874Z","iopub.execute_input":"2022-07-10T22:16:39.372158Z","iopub.status.idle":"2022-07-10T22:16:39.381861Z","shell.execute_reply.started":"2022-07-10T22:16:39.372112Z","shell.execute_reply":"2022-07-10T22:16:39.380139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ntop_tweet_bigrams=get_top_tweet_bigrams(train['text'])[:10]\nx,y=map(list,zip(*top_tweet_bigrams))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:39.383460Z","iopub.execute_input":"2022-07-10T22:16:39.384272Z","iopub.status.idle":"2022-07-10T22:16:40.506102Z","shell.execute_reply.started":"2022-07-10T22:16:39.384144Z","shell.execute_reply":"2022-07-10T22:16:40.505123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will need lot of cleaning here..","metadata":{}},{"cell_type":"markdown","source":"## Data Cleaning\nAs we know,twitter tweets always have to be cleaned before we go onto modelling.So we will do some basic cleaning such as spelling correction,removing punctuations,removing html tags and emojis etc.So let's start.","metadata":{}},{"cell_type":"code","source":"tweet = pd.concat([train,test], axis=0)\ntweet.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.507574Z","iopub.execute_input":"2022-07-10T22:16:40.507883Z","iopub.status.idle":"2022-07-10T22:16:40.534563Z","shell.execute_reply.started":"2022-07-10T22:16:40.507835Z","shell.execute_reply":"2022-07-10T22:16:40.533795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Thanks to https://www.kaggle.com/rftexas/text-only-kfold-bert\nabbreviations = {\n    \"$\" : \" dollar \",\n    \"€\" : \" euro \",\n    \"4ao\" : \"for adults only\",\n    \"a.m\" : \"before midday\",\n    \"a3\" : \"anytime anywhere anyplace\",\n    \"aamof\" : \"as a matter of fact\",\n    \"acct\" : \"account\",\n    \"adih\" : \"another day in hell\",\n    \"afaic\" : \"as far as i am concerned\",\n    \"afaict\" : \"as far as i can tell\",\n    \"afaik\" : \"as far as i know\",\n    \"afair\" : \"as far as i remember\",\n    \"afk\" : \"away from keyboard\",\n    \"app\" : \"application\",\n    \"approx\" : \"approximately\",\n    \"apps\" : \"applications\",\n    \"asap\" : \"as soon as possible\",\n    \"asl\" : \"age, sex, location\",\n    \"atk\" : \"at the keyboard\",\n    \"ave.\" : \"avenue\",\n    \"aymm\" : \"are you my mother\",\n    \"ayor\" : \"at your own risk\", \n    \"b&b\" : \"bed and breakfast\",\n    \"b+b\" : \"bed and breakfast\",\n    \"b.c\" : \"before christ\",\n    \"b2b\" : \"business to business\",\n    \"b2c\" : \"business to customer\",\n      \"b4\" : \"before\",\n    \"b4n\" : \"bye for now\",\n    \"b@u\" : \"back at you\",\n    \"bae\" : \"before anyone else\",\n    \"bak\" : \"back at keyboard\",\n    \"bbbg\" : \"bye bye be good\",\n    \"bbc\" : \"british broadcasting corporation\",\n    \"bbias\" : \"be back in a second\",\n    \"bbl\" : \"be back later\",\n    \"bbs\" : \"be back soon\",\n    \"be4\" : \"before\",\n    \"bfn\" : \"bye for now\",\n    \"blvd\" : \"boulevard\",\n    \"bout\" : \"about\",\n    \"brb\" : \"be right back\",\n    \"bros\" : \"brothers\",\n    \"brt\" : \"be right there\",\n    \"bsaaw\" : \"big smile and a wink\",\n    \"btw\" : \"by the way\",\n    \"bwl\" : \"bursting with laughter\",\n    \"c/o\" : \"care of\",\n    \"cet\" : \"central european time\",\n    \"cf\" : \"compare\",\n    \"cia\" : \"central intelligence agency\",\n    \"csl\" : \"can not stop laughing\",\n    \"cu\" : \"see you\",\n    \"cul8r\" : \"see you later\",\n    \"cv\" : \"curriculum vitae\",\n    \"cwot\" : \"complete waste of time\",\n    \"cya\" : \"see you\",\n    \"cyt\" : \"see you tomorrow\",\n    \"dae\" : \"does anyone else\",\n    \"dbmib\" : \"do not bother me i am busy\",\n    \"diy\" : \"do it yourself\",\n    \"dm\" : \"direct message\",\n    \"dwh\" : \"during work hours\",\n    \"e123\" : \"easy as one two three\",\n    \"eet\" : \"eastern european time\",\n    \"eg\" : \"example\",\n    \"embm\" : \"early morning business meeting\",\n    \"encl\" : \"enclosed\",\n    \"encl.\" : \"enclosed\",\n    \"etc\" : \"and so on\",\n    \"faq\" : \"frequently asked questions\",\n    \"fawc\" : \"for anyone who cares\",\n    \"fb\" : \"facebook\",\n    \"fc\" : \"fingers crossed\",\n    \"fig\" : \"figure\",\n    \"fimh\" : \"forever in my heart\", \n    \"ft.\" : \"feet\",\n    \"ft\" : \"featuring\",\n    \"ftl\" : \"for the loss\",\n    \"ftw\" : \"for the win\",\n    \"fwiw\" : \"for what it is worth\",\n    \"fyi\" : \"for your information\",\n    \"g9\" : \"genius\",\n    \"gahoy\" : \"get a hold of yourself\",\n    \"gal\" : \"get a life\",\n    \"gcse\" : \"general certificate of secondary education\",\n    \"gfn\" : \"gone for now\",\n    \"gg\" : \"good game\",\n    \"gl\" : \"good luck\",\n    \"glhf\" : \"good luck have fun\",\n    \"gmt\" : \"greenwich mean time\",\n    \"gmta\" : \"great minds think alike\",\n    \"gn\" : \"good night\",\n    \"g.o.a.t\" : \"greatest of all time\",\n    \"goat\" : \"greatest of all time\",\n    \"goi\" : \"get over it\",\n    \"gps\" : \"global positioning system\",\n    \"gr8\" : \"great\",\n    \"gratz\" : \"congratulations\",\n    \"gyal\" : \"girl\",\n    \"h&c\" : \"hot and cold\",\n    \"hp\" : \"horsepower\",\n    \"hr\" : \"hour\",\n    \"hrh\" : \"his royal highness\",\n    \"ht\" : \"height\",\n    \"ibrb\" : \"i will be right back\",\n    \"ic\" : \"i see\",\n    \"icq\" : \"i seek you\",\n    \"icymi\" : \"in case you missed it\",\n    \"idc\" : \"i do not care\",\n    \"idgadf\" : \"i do not give a damn fuck\",\n    \"idgaf\" : \"i do not give a fuck\",\n    \"idk\" : \"i do not know\",\n    \"ie\" : \"that is\",\n    \"i.e\" : \"that is\",\n    \"ifyp\" : \"i feel your pain\",\n    \"IG\" : \"instagram\",\n    \"iirc\" : \"if i remember correctly\",\n    \"ilu\" : \"i love you\",\n    \"ily\" : \"i love you\",\n    \"imho\" : \"in my humble opinion\",\n    \"imo\" : \"in my opinion\",\n    \"imu\" : \"i miss you\",\n    \"iow\" : \"in other words\",\n    \"irl\" : \"in real life\",\n    \"j4f\" : \"just for fun\",\n    \"jic\" : \"just in case\",\n    \"jk\" : \"just kidding\",\n    \"jsyk\" : \"just so you know\",\n    \"l8r\" : \"later\",\n    \"lb\" : \"pound\",\n    \"lbs\" : \"pounds\",\n    \"ldr\" : \"long distance relationship\",\n    \"lmao\" : \"laugh my ass off\",\n    \"lmfao\" : \"laugh my fucking ass off\",\n    \"lol\" : \"laughing out loud\",\n    \"ltd\" : \"limited\",\n    \"ltns\" : \"long time no see\",\n    \"m8\" : \"mate\",\n    \"mf\" : \"motherfucker\",\n    \"mfs\" : \"motherfuckers\",\n    \"mfw\" : \"my face when\",\n    \"mofo\" : \"motherfucker\",\n    \"mph\" : \"miles per hour\",\n    \"mr\" : \"mister\",\n    \"mrw\" : \"my reaction when\",\n    \"ms\" : \"miss\",\n    \"mte\" : \"my thoughts exactly\",\n    \"nagi\" : \"not a good idea\",\n    \"nbc\" : \"national broadcasting company\",\n    \"nbd\" : \"not big deal\",\n    \"nfs\" : \"not for sale\",\n    \"ngl\" : \"not going to lie\",\n    \"nhs\" : \"national health service\",\n    \"nrn\" : \"no reply necessary\",\n    \"nsfl\" : \"not safe for life\",\n    \"nsfw\" : \"not safe for work\",\n    \"nth\" : \"nice to have\",\n    \"nvr\" : \"never\",\n    \"nyc\" : \"new york city\",\n    \"oc\" : \"original content\",\n    \"og\" : \"original\",\n    \"ohp\" : \"overhead projector\",\n    \"oic\" : \"oh i see\",\n    \"omdb\" : \"over my dead body\",\n    \"omg\" : \"oh my god\",\n    \"omw\" : \"on my way\",\n    \"p.a\" : \"per annum\",\n    \"p.m\" : \"after midday\",\n    \"pm\" : \"prime minister\",\n    \"poc\" : \"people of color\",\n    \"pov\" : \"point of view\",\n    \"pp\" : \"pages\",\n    \"ppl\" : \"people\",\n    \"prw\" : \"parents are watching\",\n    \"ps\" : \"postscript\",\n    \"pt\" : \"point\",\n    \"ptb\" : \"please text back\",\n    \"pto\" : \"please turn over\",\n    \"qpsa\" : \"what happens\", #\"que pasa\",\n    \"ratchet\" : \"rude\",\n    \"rbtl\" : \"read between the lines\",\n    \"rlrt\" : \"real life retweet\", \n    \"rofl\" : \"rolling on the floor laughing\",\n    \"roflol\" : \"rolling on the floor laughing out loud\",\n    \"rotflmao\" : \"rolling on the floor laughing my ass off\",\n    \"rt\" : \"retweet\",\n    \"ruok\" : \"are you ok\",\n    \"sfw\" : \"safe for work\",\n    \"sk8\" : \"skate\",\n    \"smh\" : \"shake my head\",\n    \"sq\" : \"square\",\n    \"srsly\" : \"seriously\", \n    \"ssdd\" : \"same stuff different day\",\n    \"tbh\" : \"to be honest\",\n     \"tbs\" : \"tablespooful\",\n    \"tbsp\" : \"tablespooful\",\n    \"tfw\" : \"that feeling when\",\n    \"thks\" : \"thank you\",\n    \"tho\" : \"though\",\n    \"thx\" : \"thank you\",\n    \"tia\" : \"thanks in advance\",\n    \"til\" : \"today i learned\",\n    \"tl;dr\" : \"too long i did not read\",\n    \"tldr\" : \"too long i did not read\",\n    \"tmb\" : \"tweet me back\",\n    \"tntl\" : \"trying not to laugh\",\n    \"ttyl\" : \"talk to you later\",\n    \"u\" : \"you\",\n    \"u2\" : \"you too\",\n    \"u4e\" : \"yours for ever\",\n    \"utc\" : \"coordinated universal time\",\n    \"w/\" : \"with\",\n    \"w/o\" : \"without\",\n    \"w8\" : \"wait\",\n    \"wassup\" : \"what is up\",\n    \"wb\" : \"welcome back\",\n    \"wtf\" : \"what the fuck\",\n    \"wtg\" : \"way to go\",\n    \"wtpa\" : \"where the party at\",\n    \"wuf\" : \"where are you from\",\n    \"wuzup\" : \"what is up\",\n    \"wywh\" : \"wish you were here\",\n    \"yd\" : \"yard\",\n    \"ygtr\" : \"you got that right\",\n    \"ynk\" : \"you never know\",\n    \"zzz\" : \"sleeping bored and tired\"\n}\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.538305Z","iopub.execute_input":"2022-07-10T22:16:40.541689Z","iopub.status.idle":"2022-07-10T22:16:40.588408Z","shell.execute_reply.started":"2022-07-10T22:16:40.541608Z","shell.execute_reply":"2022-07-10T22:16:40.587346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Thanks to https://www.kaggle.com/rftexas/text-only-kfold-bert\ndef convert_abbrev(word):\n    return abbreviations[word.lower()] if word.lower() in abbreviations.keys() else word","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.589908Z","iopub.execute_input":"2022-07-10T22:16:40.590265Z","iopub.status.idle":"2022-07-10T22:16:40.604946Z","shell.execute_reply.started":"2022-07-10T22:16:40.590211Z","shell.execute_reply":"2022-07-10T22:16:40.602367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_abbrev_in_text(text):\n    tokens = word_tokenize(text)\n    tokens = [convert_abbrev(word) for word in tokens]\n    text = ' '.join(tokens)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.606424Z","iopub.execute_input":"2022-07-10T22:16:40.606950Z","iopub.status.idle":"2022-07-10T22:16:40.622945Z","shell.execute_reply.started":"2022-07-10T22:16:40.606896Z","shell.execute_reply":"2022-07-10T22:16:40.621075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r\"https?:\\/\\/t.co\\/[A-Za-z0-9]+\")\n    return url.sub(\"\",text)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.624487Z","iopub.execute_input":"2022-07-10T22:16:40.624846Z","iopub.status.idle":"2022-07-10T22:16:40.630486Z","shell.execute_reply.started":"2022-07-10T22:16:40.624781Z","shell.execute_reply":"2022-07-10T22:16:40.629471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet['text'] = tweet['text'].apply(remove_URL)\ntweet[\"text\"] = tweet[\"text\"].apply(lambda x: convert_abbrev_in_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:40.632203Z","iopub.execute_input":"2022-07-10T22:16:40.632846Z","iopub.status.idle":"2022-07-10T22:16:44.014290Z","shell.execute_reply.started":"2022-07-10T22:16:40.632796Z","shell.execute_reply":"2022-07-10T22:16:44.013076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.corpus import stopwords\n\n#function for removing pattern\ndef remove_pattern(input_txt, pattern):\n    r = re.findall(pattern, input_txt)\n    for i in r:\n        input_txt = re.sub(i, '', input_txt)\n    return input_txt\n\n\ndef remove_not_ASCII(text):\n    text = ''.join([word for word in text if word in string.printable])\n    return text\n\n\n\n# remove '#' handle\ntweet['tweet'] = np.vectorize(remove_pattern)(tweet['text'], \"#[\\w]*\")\n\n# Remove non printable characters\ntweet['tweet'] = tweet['tweet'].apply(remove_not_ASCII)\n\n#Delete everything except alphabet\ntweet['tweet'] = tweet['tweet'].str.replace(\"[^a-zA-Z#]\", \" \")\n\n#Dropping words whose length is less than 3\ntweet['tweet'] = tweet['tweet'].apply(lambda x: ' '.join([w for w in x.split() if len(w)>3]))\n\n#convert all the words into lower case\ntweet['tweet'] = tweet['tweet'].str.lower()\n\n\nset(stopwords.words('english'))\nstops = set(stopwords.words('english')) \n\n# tokens of words  \ntweet['tokenized_sents'] = tweet.apply(lambda row: nltk.word_tokenize(row['tweet']), axis=1)\n\n#function to remove stop words\ndef remove_stops(row):\n    my_list = row['tokenized_sents']\n    meaningful_words = [w for w in my_list if not w in stops]\n    return (meaningful_words)\n\n#removing stop words\ntweet['clean_tweet'] = tweet.apply(remove_stops, axis=1)\ntweet.drop([\"tweet\",\"tokenized_sents\"], axis = 1, inplace = True)\n\n\ndef rejoin_words(row):\n    my_list = row['clean_tweet']\n    joined_words = ( \" \".join(my_list))\n    return joined_words\n\ntweet['clean_join_tweet'] = tweet.apply(rejoin_words, axis=1)\ntweet.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:44.015827Z","iopub.execute_input":"2022-07-10T22:16:44.016159Z","iopub.status.idle":"2022-07-10T22:16:49.237453Z","shell.execute_reply.started":"2022-07-10T22:16:44.016050Z","shell.execute_reply":"2022-07-10T22:16:49.236673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Most common words\n\nfrom wordcloud import WordCloud \nall_word = ' '.join([text for text in tweet['clean_join_tweet']])\nwordcloud = WordCloud(width=800, height=500, random_state=21, max_font_size=110).generate(all_word) \nplt.figure(figsize=(10, 7)) \nplt.imshow(wordcloud, interpolation=\"bilinear\")\nplt.axis('off') \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:49.238763Z","iopub.execute_input":"2022-07-10T22:16:49.239035Z","iopub.status.idle":"2022-07-10T22:16:50.897400Z","shell.execute_reply.started":"2022-07-10T22:16:49.238990Z","shell.execute_reply":"2022-07-10T22:16:50.896741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = tweet.iloc[:train.shape[0],:]\ntest = tweet.iloc[train.shape[0]:,:]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:50.898633Z","iopub.execute_input":"2022-07-10T22:16:50.899025Z","iopub.status.idle":"2022-07-10T22:16:50.909159Z","shell.execute_reply.started":"2022-07-10T22:16:50.898978Z","shell.execute_reply":"2022-07-10T22:16:50.908157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Word2Vec","metadata":{}},{"cell_type":"code","source":"from gensim.models import Word2Vec","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:50.910806Z","iopub.execute_input":"2022-07-10T22:16:50.911533Z","iopub.status.idle":"2022-07-10T22:16:50.915674Z","shell.execute_reply.started":"2022-07-10T22:16:50.911328Z","shell.execute_reply":"2022-07-10T22:16:50.914745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EMBEDDING_DIM = 300","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:50.917083Z","iopub.execute_input":"2022-07-10T22:16:50.917679Z","iopub.status.idle":"2022-07-10T22:16:50.925996Z","shell.execute_reply.started":"2022-07-10T22:16:50.917606Z","shell.execute_reply":"2022-07-10T22:16:50.925117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nw2v_model = Word2Vec(sentences=train['clean_tweet'], size=EMBEDDING_DIM, window=7, min_count=1,sg=1)\nw2v_model.train(train['clean_tweet'], total_examples=len(train['clean_tweet']), epochs=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:16:50.932432Z","iopub.execute_input":"2022-07-10T22:16:50.932673Z","iopub.status.idle":"2022-07-10T22:17:02.898176Z","shell.execute_reply.started":"2022-07-10T22:16:50.932607Z","shell.execute_reply":"2022-07-10T22:17:02.897309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(w2v_model.wv.vocab)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:02.899600Z","iopub.execute_input":"2022-07-10T22:17:02.899883Z","iopub.status.idle":"2022-07-10T22:17:02.904970Z","shell.execute_reply.started":"2022-07-10T22:17:02.899843Z","shell.execute_reply":"2022-07-10T22:17:02.904172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w2v_model.most_similar('dead', topn=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:02.906497Z","iopub.execute_input":"2022-07-10T22:17:02.907107Z","iopub.status.idle":"2022-07-10T22:17:02.941556Z","shell.execute_reply.started":"2022-07-10T22:17:02.907055Z","shell.execute_reply":"2022-07-10T22:17:02.940562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Glove","metadata":{}},{"cell_type":"code","source":"!pip install  -q glove_python","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:02.943287Z","iopub.execute_input":"2022-07-10T22:17:02.948764Z","iopub.status.idle":"2022-07-10T22:17:07.933807Z","shell.execute_reply.started":"2022-07-10T22:17:02.948691Z","shell.execute_reply":"2022-07-10T22:17:07.932814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glove import Corpus, Glove","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:07.935726Z","iopub.execute_input":"2022-07-10T22:17:07.936070Z","iopub.status.idle":"2022-07-10T22:17:07.941497Z","shell.execute_reply.started":"2022-07-10T22:17:07.936017Z","shell.execute_reply":"2022-07-10T22:17:07.940548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#Creating a corpus object\ncorpus = Corpus() \n\n#Training the corpus to generate the co-occurrence matrix which is used in GloVe\ncorpus.fit(train['clean_tweet'], window=7)\n\nglove = Glove(no_components=EMBEDDING_DIM, learning_rate=0.2) \nglove.fit(corpus.matrix, epochs=30)\nglove.add_dictionary(corpus.dictionary)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:07.942969Z","iopub.execute_input":"2022-07-10T22:17:07.943566Z","iopub.status.idle":"2022-07-10T22:17:28.128436Z","shell.execute_reply.started":"2022-07-10T22:17:07.943510Z","shell.execute_reply":"2022-07-10T22:17:28.127761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(corpus.dictionary)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.131414Z","iopub.execute_input":"2022-07-10T22:17:28.131674Z","iopub.status.idle":"2022-07-10T22:17:28.138256Z","shell.execute_reply.started":"2022-07-10T22:17:28.131608Z","shell.execute_reply":"2022-07-10T22:17:28.137444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove.most_similar('dead', number=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.139542Z","iopub.execute_input":"2022-07-10T22:17:28.140046Z","iopub.status.idle":"2022-07-10T22:17:28.181068Z","shell.execute_reply.started":"2022-07-10T22:17:28.139995Z","shell.execute_reply":"2022-07-10T22:17:28.180167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenizer","metadata":{}},{"cell_type":"code","source":"# Tokenizing Text -> Repsesenting each word by a number\n# Mapping of orginal word to number is preserved in word_index property of tokenizer\n\n#Tokenized applies basic processing like changing it yo lower case, explicitely setting that as False\ntokenizer = Tokenizer()\ntokenizer.fit_on_texts(train['clean_tweet'])\n\nX = tokenizer.texts_to_sequences(train['clean_tweet'])\ntest = tokenizer.texts_to_sequences(test['clean_tweet'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.185723Z","iopub.execute_input":"2022-07-10T22:17:28.188183Z","iopub.status.idle":"2022-07-10T22:17:28.395167Z","shell.execute_reply.started":"2022-07-10T22:17:28.188115Z","shell.execute_reply":"2022-07-10T22:17:28.394290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[0][:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.396908Z","iopub.execute_input":"2022-07-10T22:17:28.397267Z","iopub.status.idle":"2022-07-10T22:17:28.403402Z","shell.execute_reply.started":"2022-07-10T22:17:28.397213Z","shell.execute_reply":"2022-07-10T22:17:28.402659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index = tokenizer.word_index\nfor word, num in word_index.items():\n    print(f\"{word} -> {num}\")\n    if num == 10:\n        break  ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.404760Z","iopub.execute_input":"2022-07-10T22:17:28.405195Z","iopub.status.idle":"2022-07-10T22:17:28.414516Z","shell.execute_reply.started":"2022-07-10T22:17:28.405144Z","shell.execute_reply":"2022-07-10T22:17:28.413751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check hist with tweets len\nplt.hist([len(x) for x in X], bins=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.415821Z","iopub.execute_input":"2022-07-10T22:17:28.416366Z","iopub.status.idle":"2022-07-10T22:17:28.683294Z","shell.execute_reply.started":"2022-07-10T22:17:28.416293Z","shell.execute_reply":"2022-07-10T22:17:28.682422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#7550 tweets have len < 7613\nnos = np.array([len(x) for x in X])\nlen(nos[nos<15]), nos.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.684571Z","iopub.execute_input":"2022-07-10T22:17:28.685005Z","iopub.status.idle":"2022-07-10T22:17:28.695678Z","shell.execute_reply.started":"2022-07-10T22:17:28.684956Z","shell.execute_reply":"2022-07-10T22:17:28.694945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"maxlen = 15\n\nX = pad_sequences(X, maxlen=maxlen)\ntest = pad_sequences(test, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.696917Z","iopub.execute_input":"2022-07-10T22:17:28.697353Z","iopub.status.idle":"2022-07-10T22:17:28.802891Z","shell.execute_reply.started":"2022-07-10T22:17:28.697303Z","shell.execute_reply":"2022-07-10T22:17:28.802142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding 1 because of reserved 0 index\n# Embedding Layer creates one more vector for \"UNKNOWN\" words, or padded words (0s). This Vector is filled with zeros.\n# Thus our vocab size inceeases by 1\nvocab_size = len(tokenizer.word_index) + 1","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.804213Z","iopub.execute_input":"2022-07-10T22:17:28.804490Z","iopub.status.idle":"2022-07-10T22:17:28.808138Z","shell.execute_reply.started":"2022-07-10T22:17:28.804446Z","shell.execute_reply":"2022-07-10T22:17:28.807433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_weight_matrix_w2v(model, vocab):\n    # total vocabulary size plus 0 for unknown words\n    vocab_size = len(vocab) + 1\n    # define weight matrix dimensions with all 0\n    weight_matrix = np.zeros((vocab_size, EMBEDDING_DIM))\n    # step vocab, store vectors using the Tokenizer's integer mapping\n    for word, i in vocab.items():\n        weight_matrix[i] = model[word]\n    return weight_matrix\n\ndef get_weight_matrix_glove(model, vocab):\n    # total vocabulary size plus 0 for unknown words\n    vocab_size = len(vocab) + 1\n    # define weight matrix dimensions with all 0\n    weight_matrix = np.zeros((vocab_size, EMBEDDING_DIM))\n    # step vocab, store vectors using the Tokenizer's integer mapping\n    for word, i in vocab.items():\n        weight_matrix[i] = model.word_vectors[model.dictionary[word]]\n    return weight_matrix\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.809513Z","iopub.execute_input":"2022-07-10T22:17:28.809943Z","iopub.status.idle":"2022-07-10T22:17:28.820996Z","shell.execute_reply.started":"2022-07-10T22:17:28.809898Z","shell.execute_reply":"2022-07-10T22:17:28.820347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_vectors_w2v = get_weight_matrix_w2v(w2v_model, word_index)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.822836Z","iopub.execute_input":"2022-07-10T22:17:28.823404Z","iopub.status.idle":"2022-07-10T22:17:28.971361Z","shell.execute_reply.started":"2022-07-10T22:17:28.823126Z","shell.execute_reply":"2022-07-10T22:17:28.970622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_vectors_glove = get_weight_matrix_glove(glove, word_index)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:28.973629Z","iopub.execute_input":"2022-07-10T22:17:28.974156Z","iopub.status.idle":"2022-07-10T22:17:29.024983Z","shell.execute_reply.started":"2022-07-10T22:17:28.974104Z","shell.execute_reply":"2022-07-10T22:17:29.024107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baseline Model","metadata":{}},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense,Embedding,LSTM,Dropout,Bidirectional,GRU\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:29.026507Z","iopub.execute_input":"2022-07-10T22:17:29.026966Z","iopub.status.idle":"2022-07-10T22:17:29.032289Z","shell.execute_reply.started":"2022-07-10T22:17:29.026781Z","shell.execute_reply":"2022-07-10T22:17:29.031281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(X, train['target'], test_size=0.2, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:29.033914Z","iopub.execute_input":"2022-07-10T22:17:29.034450Z","iopub.status.idle":"2022-07-10T22:17:29.044629Z","shell.execute_reply.started":"2022-07-10T22:17:29.034163Z","shell.execute_reply":"2022-07-10T22:17:29.043991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Defining Neural Network\nmodel = Sequential()\n#Non-trainable embedding layer\nmodel.add(Embedding(vocab_size, \n                    output_dim=EMBEDDING_DIM, \n                    weights=[embedding_vectors_w2v], \n                    input_length=maxlen, \n                    trainable=False))\n#LSTM \n\nmodel.add(Bidirectional(LSTM(units=200, recurrent_dropout=0.2, dropout=0.2, return_sequences=False)))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(optimizer=keras.optimizers.Adam(lr=0.005), loss='binary_crossentropy', metrics=['acc'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:29.046446Z","iopub.execute_input":"2022-07-10T22:17:29.047026Z","iopub.status.idle":"2022-07-10T22:17:29.734445Z","shell.execute_reply.started":"2022-07-10T22:17:29.046747Z","shell.execute_reply":"2022-07-10T22:17:29.733507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:29.736006Z","iopub.execute_input":"2022-07-10T22:17:29.736291Z","iopub.status.idle":"2022-07-10T22:17:29.747067Z","shell.execute_reply.started":"2022-07-10T22:17:29.736244Z","shell.execute_reply":"2022-07-10T22:17:29.744235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x_train, \n                    y_train, \n                    batch_size=64, \n                    validation_data=(x_test,y_test), \n                    epochs=10, \n                    verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:29.748653Z","iopub.execute_input":"2022-07-10T22:17:29.748907Z","iopub.status.idle":"2022-07-10T22:17:59.818651Z","shell.execute_reply.started":"2022-07-10T22:17:29.748862Z","shell.execute_reply":"2022-07-10T22:17:59.817683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = [i for i in range(10)]\nfig , ax = plt.subplots(1,2)\ntrain_acc = history.history['acc']\ntrain_loss = history.history['loss']\nval_acc = history.history['val_acc']\nval_loss = history.history['val_loss']\nfig.set_size_inches(20,10)\n\nax[0].plot(epochs , train_acc , 'go-' , label = 'Training Accuracy')\nax[0].plot(epochs , val_acc , 'ro-' , label = 'Testing Accuracy')\nax[0].set_title('Training & Testing Accuracy')\nax[0].legend()\nax[0].set_xlabel(\"Epochs\")\nax[0].set_ylabel(\"Accuracy\")\n\nax[1].plot(epochs , train_loss , 'go-' , label = 'Training Loss')\nax[1].plot(epochs , val_loss , 'ro-' , label = 'Testing Loss')\nax[1].set_title('Training & Testing Loss')\nax[1].legend()\nax[1].set_xlabel(\"Epochs\")\nax[1].set_ylabel(\"Loss\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:17:59.820061Z","iopub.execute_input":"2022-07-10T22:17:59.820376Z","iopub.status.idle":"2022-07-10T22:18:00.374233Z","shell.execute_reply.started":"2022-07-10T22:17:59.820325Z","shell.execute_reply":"2022-07-10T22:18:00.373529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Defining Neural Network\nmodel_glove = Sequential()\n#Non-trainable embedding layer\nmodel_glove.add(Embedding(vocab_size, \n                    output_dim=EMBEDDING_DIM, \n                    weights=[embedding_vectors_glove], \n                    input_length=maxlen, \n                    trainable=False))\n#LSTM \n\nmodel_glove.add(Bidirectional(LSTM(units=200, recurrent_dropout=0.2, dropout=0.2, return_sequences=False)))\nmodel_glove.add(Dense(1, activation='sigmoid'))\nmodel_glove.compile(optimizer=keras.optimizers.Adam(lr=0.005), loss='binary_crossentropy', metrics=['acc'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:00.376566Z","iopub.execute_input":"2022-07-10T22:18:00.377028Z","iopub.status.idle":"2022-07-10T22:18:01.036944Z","shell.execute_reply.started":"2022-07-10T22:18:00.376977Z","shell.execute_reply":"2022-07-10T22:18:01.036211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_glove.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:01.038323Z","iopub.execute_input":"2022-07-10T22:18:01.038602Z","iopub.status.idle":"2022-07-10T22:18:01.046633Z","shell.execute_reply.started":"2022-07-10T22:18:01.038556Z","shell.execute_reply":"2022-07-10T22:18:01.045753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_glove.fit(x_train, \n                    y_train, \n                    batch_size=64, \n                    validation_data=(x_test,y_test), \n                    epochs=10, \n                    verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:01.047965Z","iopub.execute_input":"2022-07-10T22:18:01.048403Z","iopub.status.idle":"2022-07-10T22:18:31.399295Z","shell.execute_reply.started":"2022-07-10T22:18:01.048356Z","shell.execute_reply":"2022-07-10T22:18:31.398653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = [i for i in range(10)]\nfig , ax = plt.subplots(1,2)\ntrain_acc = history.history['acc']\ntrain_loss = history.history['loss']\nval_acc = history.history['val_acc']\nval_loss = history.history['val_loss']\nfig.set_size_inches(20,10)\n\nax[0].plot(epochs , train_acc , 'go-' , label = 'Training Accuracy')\nax[0].plot(epochs , val_acc , 'ro-' , label = 'Testing Accuracy')\nax[0].set_title('Training & Testing Accuracy')\nax[0].legend()\nax[0].set_xlabel(\"Epochs\")\nax[0].set_ylabel(\"Accuracy\")\n\nax[1].plot(epochs , train_loss , 'go-' , label = 'Training Loss')\nax[1].plot(epochs , val_loss , 'ro-' , label = 'Testing Loss')\nax[1].set_title('Training & Testing Loss')\nax[1].legend()\nax[1].set_xlabel(\"Epochs\")\nax[1].set_ylabel(\"Loss\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:31.400661Z","iopub.execute_input":"2022-07-10T22:18:31.400948Z","iopub.status.idle":"2022-07-10T22:18:31.945657Z","shell.execute_reply.started":"2022-07-10T22:18:31.400900Z","shell.execute_reply":"2022-07-10T22:18:31.944868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Both models perform bad at validation and need improvement. This will happen in future notebook versions, but now we can notice that Word2Vec embeddings give better results than Glove embeddings**","metadata":{}},{"cell_type":"markdown","source":"## Making our submission","metadata":{}},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:31.947050Z","iopub.execute_input":"2022-07-10T22:18:31.947518Z","iopub.status.idle":"2022-07-10T22:18:31.957502Z","shell.execute_reply.started":"2022-07-10T22:18:31.947470Z","shell.execute_reply":"2022-07-10T22:18:31.956742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pre=model.predict(test)\ny_pre=np.round(y_pre).astype(int).reshape(3263)\nsub=pd.DataFrame({'id':sample_sub['id'].values.tolist(),'target':y_pre})\nsub.to_csv('submission.csv',index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:31.959287Z","iopub.execute_input":"2022-07-10T22:18:31.959736Z","iopub.status.idle":"2022-07-10T22:18:32.758161Z","shell.execute_reply.started":"2022-07-10T22:18:31.959558Z","shell.execute_reply":"2022-07-10T22:18:32.757354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T22:18:32.760420Z","iopub.execute_input":"2022-07-10T22:18:32.760681Z","iopub.status.idle":"2022-07-10T22:18:32.772360Z","shell.execute_reply.started":"2022-07-10T22:18:32.760628Z","shell.execute_reply":"2022-07-10T22:18:32.771551Z"},"trusted":true},"execution_count":null,"outputs":[]}]}